{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceType":"competition","sourceId":4117,"databundleVersionId":46665},{"sourceType":"datasetVersion","sourceId":15863312,"datasetId":10170054,"databundleVersionId":16815445},{"sourceType":"kernelVersion","sourceId":313406024}],"dockerImageVersionId":31329,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Phase 2+3 — CNN Feature Extraction + XGBoost Training\n\n**Input từ Phase 1:**\n```\n/kaggle/input/notebooks/nhatnguyenhcmusk24/nahhhhhhh/features/\n  ├── image_features.zip      (10,868 × 256×256×3 .npy)\n  ├── tabular_features.csv    (10,868 × 525 opcode features + Class)\n  └── vocabulary.pkl          (top_uni + top_bi)\n```\n\n**Pipeline Phase 2+3:**\n```\nimage_features.zip → extract → EfficientNet-B0 (frozen) → cnn_features (1280-d)\ntabular_features.csv                                     → ngram_features (525-d)\n7z l train.7z + test.7z                                  → metadata (4-d)\ntest.7z → batch extract → Markov + opcode               → test features\n                                                              ↓\nCONCAT → Feature Selection (RF) → XGBoost (Stratified 5-Fold + sample_weight)\n       → Pseudo-labeling → Ensemble → submission.csv\n```","metadata":{}},{"cell_type":"code","source":"import sys, os, gc, time, logging, zipfile, shutil, subprocess, pickle\nimport numpy as np\nimport pandas as pd\nimport scipy.sparse as sp\nfrom pathlib import Path\nfrom tqdm.auto import tqdm\n\nimport torch\nimport torch.nn as nn\nfrom torch.utils.data import Dataset, DataLoader\nimport torchvision.models as models\nimport torchvision.transforms as T\n\nimport xgboost as xgb\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.utils.class_weight import compute_sample_weight\nfrom sklearn.metrics import accuracy_score, log_loss\n\n# ── Utils ─────────────────────────────────────────────────────────────────────\nUTILS_PATH = Path('/kaggle/input/datasets/boiboii326/malware')\nsys.path.append(str(UTILS_PATH))\nfrom malware_utils_v2 import Config, GPU, FileOps, ImageOps, OpcodeOps, CacheOps\n\n# ── Logging ───────────────────────────────────────────────────────────────────\nlogging.basicConfig(\n    level=logging.INFO,\n    format='%(asctime)s [%(levelname)s] %(message)s',\n    datefmt='%H:%M:%S',\n    handlers=[\n        logging.StreamHandler(sys.stdout),\n        logging.FileHandler('/kaggle/working/phase2.log')\n    ]\n)\nlog = logging.getLogger('phase2')\n\n# ── Paths ─────────────────────────────────────────────────────────────────────\nPHASE1_DIR  = Path('/kaggle/input/notebooks/boiboii326/tempp/features')\nCOMP_DIR    = Path('/kaggle/input/competitions/malware-classification')\nWORK_DIR    = Path('/kaggle/working')\nIMG_DIR     = WORK_DIR / 'train_images'   # sau khi extract zip\nFEAT_DIR    = WORK_DIR / 'features'\nSCRATCH     = WORK_DIR / 'scratch'\n\nfor d in [IMG_DIR, FEAT_DIR, SCRATCH]:\n    d.mkdir(parents=True, exist_ok=True)\n\n# ── Config ────────────────────────────────────────────────────────────────────\ncfg = Config(base_dir=COMP_DIR)\nGPU.setup()\nDEVICE = torch.device('cuda' if torch.cuda.is_available() else 'cpu')\n\nSEED     = 42\nN_FOLDS  = 5\nN_CLASSES= 9\n\nCLASS_NAMES = {\n    1:'Ramnit', 2:'Lollipop', 3:'Kelihos_ver3', 4:'Vundo', 5:'Simda',\n    6:'Tracur', 7:'Kelihos_ver1', 8:'Obfuscator.ACY', 9:'Gatak'\n}\n\nlog.info('Device: %s', DEVICE)\nlog.info('Phase1 dir: %s (exists=%s)', PHASE1_DIR, PHASE1_DIR.exists())\nlog.info('Disk free: %.1f GB', FileOps.get_free_gb())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-22T05:29:28.417360Z","iopub.execute_input":"2026-04-22T05:29:28.417732Z","iopub.status.idle":"2026-04-22T05:29:29.794680Z","shell.execute_reply.started":"2026-04-22T05:29:28.417701Z","shell.execute_reply":"2026-04-22T05:29:29.794099Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Load Data process outputs","metadata":{}},{"cell_type":"code","source":"# ── Labels ────────────────────────────────────────────────────────────────────\ndf_labels = pd.read_csv(COMP_DIR / 'trainLabels.csv')\ndf_labels.columns = df_labels.columns.str.strip()\nfor c in df_labels.columns:\n    if c.lower() in ('id','name'): df_labels = df_labels.rename(columns={c:'Id'})\n    if c.lower() == 'class':       df_labels = df_labels.rename(columns={c:'Class'})\n\ndf_labels['label'] = df_labels['Class'] - 1\nALL_IDS = df_labels['Id'].tolist()\nlog.info('Train samples: %d', len(ALL_IDS))\n\n# ── Tabular features từ Phase 1 v2 (~850 cols) ────────────────────────────────\ndf_tab = pd.read_csv(PHASE1_DIR / 'tabular_features.csv', index_col='Id')\nlog.info('Tabular raw shape: %s', df_tab.shape)\n\n# ── Auto-detect 4 nhóm cột ────────────────────────────────────────────────────\nMETA_COLS = ['bytes_size', 'asm_size', 'size_ratio']\n\n# Phase A extra: sections, API, funcs, strings, regs\nEXTRA_COLS = [c for c in df_tab.columns\n              if c.startswith('sec_')\n              or c.startswith('api_')\n              or c.startswith('has_')\n              or c.startswith('reg_')\n              or c.startswith('n_')\n              or c in ('avg_func_size', 'proc_endp_match_ratio',\n                       'indirect_call_pct')]\n\n# Phase B byte stats (tất cả cols bắt đầu 'byte_')\nBYTE_COLS = [c for c in df_tab.columns if c.startswith('byte_')]\n\ndf_meta  = df_tab[META_COLS].copy()\ndf_extra = df_tab[EXTRA_COLS].copy()\ndf_byte  = df_tab[BYTE_COLS].copy()\ndf_tab   = df_tab.drop(columns=META_COLS + EXTRA_COLS + BYTE_COLS)\n\ny_tab = df_tab.pop('Class') if 'Class' in df_tab.columns else None\n\nlog.info('Split | Opcode=%d  Extra=%d  Byte=%d  Meta=%d',\n         df_tab.shape[1], df_extra.shape[1], df_byte.shape[1], df_meta.shape[1])\n\ntop_uni, top_bi = CacheOps.load_vocabulary(PHASE1_DIR / 'vocabulary.pkl')\nlog.info('Vocabulary: %d uni + %d bi', len(top_uni), len(top_bi))\n\nprint(f'Opcode : {df_tab.shape}')\nprint(f'Extra  : {df_extra.shape}  (Phase A: sec/api/func/string/reg)')\nprint(f'Byte   : {df_byte.shape}   (Phase B: byte stats)')\nprint(f'Meta   : {df_meta.shape}')\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-22T05:29:35.275347Z","iopub.execute_input":"2026-04-22T05:29:35.276261Z","iopub.status.idle":"2026-04-22T05:29:36.858585Z","shell.execute_reply.started":"2026-04-22T05:29:35.276230Z","shell.execute_reply":"2026-04-22T05:29:36.858043Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"##  Extract image_features.zip → train_images/","metadata":{}},{"cell_type":"code","source":"# ── Extract zip một lần, load nhanh về sau ────────────────────────────────────\nzip_path = PHASE1_DIR / 'image_features.zip'\nalready  = list(IMG_DIR.glob('*.npy'))\n\nif len(already) >= len(ALL_IDS):\n    log.info('Images đã extract: %d files — skip', len(already))\nelse:\n    log.info('Extracting %s → %s ...', zip_path.name, IMG_DIR)\n    t0 = time.time()\n    with zipfile.ZipFile(zip_path, 'r') as zf:\n        zf.extractall(IMG_DIR)\n    n = len(list(IMG_DIR.glob('*.npy')))\n    log.info('Extracted %d files in %.1fs', n, time.time()-t0)\n\nlog.info('Disk free after extract: %.1f GB', FileOps.get_free_gb())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-22T05:29:44.217178Z","iopub.execute_input":"2026-04-22T05:29:44.217735Z","iopub.status.idle":"2026-04-22T05:30:01.413934Z","shell.execute_reply.started":"2026-04-22T05:29:44.217709Z","shell.execute_reply":"2026-04-22T05:30:01.413373Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## CNN Feature Extraction (EfficientNet-B0, Frozen)","metadata":{}},{"cell_type":"code","source":"# ── Dataset ───────────────────────────────────────────────────────────────────\nclass MalwareImageDataset(Dataset):\n    \"\"\"\n    Load .npy tensors (256,256,3) uint8.\n    Normalize về ImageNet mean/std để dùng pretrained weights.\n    \"\"\"\n    def __init__(self, sample_ids: list, img_dir: Path):\n        self.ids     = sample_ids\n        self.img_dir = img_dir\n        self.tf = T.Compose([\n            T.ToTensor(),                          # HWC uint8 → CHW float32 [0,1]\n            T.Normalize(mean=[0.485, 0.456, 0.406],\n                        std =[0.229, 0.224, 0.225]),\n        ])\n\n    def __len__(self):  return len(self.ids)\n\n    def __getitem__(self, idx):\n        arr = np.load(self.img_dir / f'{self.ids[idx]}.npy')  # (256,256,3) uint8\n        return self.tf(arr), self.ids[idx]\n\n\n# ── Model: EfficientNet-B0 — bỏ head, giữ feature extractor ──────────────────\nclass CNNExtractor(nn.Module):\n    def __init__(self):\n        super().__init__()\n        backbone = models.efficientnet_b0(weights='IMAGENET1K_V1')\n        # Bỏ classifier head, giữ features + avgpool\n        self.features = backbone.features\n        self.avgpool  = backbone.avgpool   # AdaptiveAvgPool2d(1)\n        # Freeze toàn bộ\n        for p in self.parameters():\n            p.requires_grad = False\n\n    def forward(self, x):\n        x = self.features(x)\n        x = self.avgpool(x)\n        return x.flatten(1)   # → (batch, 1280)\n\n\n# ── Extract features ──────────────────────────────────────────────────────────\nCNN_FEAT_PATH = FEAT_DIR / 'cnn_features_train.npy'\nCNN_IDS_PATH  = FEAT_DIR / 'cnn_ids_train.pkl'\n\nif CNN_FEAT_PATH.exists():\n    log.info('CNN features đã có — skip extraction')\n    cnn_feat_train = np.load(CNN_FEAT_PATH)\n    with open(CNN_IDS_PATH, 'rb') as f:\n        cnn_ids_train = pickle.load(f)\nelse:\n    model = CNNExtractor().to(DEVICE).eval()\n    log.info('EfficientNet-B0 loaded (frozen) — output dim: 1280')\n\n    dataset = MalwareImageDataset(ALL_IDS, IMG_DIR)\n    loader  = DataLoader(dataset, batch_size=64, shuffle=False,\n                         num_workers=4, pin_memory=True)\n\n    all_feats = []\n    all_ids   = []\n    t0 = time.time()\n\n    with torch.no_grad():\n        for imgs, ids in tqdm(loader, desc='CNN extract', ncols=80):\n            feats = model(imgs.to(DEVICE))\n            all_feats.append(feats.cpu().numpy())\n            all_ids.extend(ids)\n\n    cnn_feat_train = np.vstack(all_feats)   # (N, 1280)\n    cnn_ids_train  = all_ids\n\n    np.save(CNN_FEAT_PATH, cnn_feat_train)\n    with open(CNN_IDS_PATH, 'wb') as f:\n        pickle.dump(cnn_ids_train, f, protocol=4)\n\n    elapsed = time.time() - t0\n    log.info('CNN extraction done: shape=%s in %.1f min',\n             cnn_feat_train.shape, elapsed/60)\n\n    del model\n    torch.cuda.empty_cache()\n    gc.collect()\n\nprint(f'CNN features shape: {cnn_feat_train.shape}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-22T05:30:14.985106Z","iopub.execute_input":"2026-04-22T05:30:14.985856Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Test set processing\n\nExtract test.7z theo batch → Markov image + Opcode features dùng vocabulary.pkl từ Phase 1","metadata":{}},{"cell_type":"code","source":"TEST_IMG_DIR    = WORK_DIR / 'test_images'\nTEST_CNN_PATH   = FEAT_DIR / 'cnn_features_test.npy'\nTEST_TAB_PATH   = FEAT_DIR / 'tabular_features_test.csv'\nTEST_EXTRA_PATH = FEAT_DIR / 'extra_features_test.csv'\nTEST_BYTE_PATH  = FEAT_DIR / 'byte_stats_test.csv'\nTEST_META_PATH  = FEAT_DIR / 'metadata_test.csv'\nTEST_IMG_DIR.mkdir(exist_ok=True)\n\ndf_sub_template = pd.read_csv(COMP_DIR / 'sampleSubmission.csv')\nTEST_IDS = df_sub_template['Id'].tolist()\nlog.info('Test samples: %d', len(TEST_IDS))\n\n# Force delete stale cache\nfor p in [TEST_CNN_PATH, TEST_TAB_PATH, TEST_EXTRA_PATH, TEST_BYTE_PATH, TEST_META_PATH]:\n    if p.exists():\n        p.unlink()\n        log.info('Deleted stale: %s', p.name)\n\nlog.info('Processing test set (v2: extra + byte stats)...')\n\ndef get_file_sizes_by_ext(archive: Path):\n    result = subprocess.run(['7z', 'l', '-ba', str(archive)],\n                            capture_output=True, text=True)\n    bs, as_ = {}, {}\n    for line in result.stdout.splitlines():\n        parts = line.strip().split()\n        if len(parts) >= 5:\n            try:\n                sz = int(parts[3])\n                p = Path(parts[-1])\n                if p.suffix == '.bytes': bs[p.stem] = sz\n                elif p.suffix == '.asm': as_[p.stem] = sz\n            except ValueError:\n                pass\n    return bs, as_\n\ndef process_test_batch(batch_ids, archive, img_dir, raw_records):\n    FileOps.extract_batch(archive, batch_ids, SCRATCH)\n    bytes_idx, asm_idx = {}, {}\n    for p in SCRATCH.rglob('*'):\n        if p.is_file():\n            if p.suffix == '.bytes': bytes_idx[p.stem] = p\n            elif p.suffix == '.asm': asm_idx[p.stem]   = p\n\n    for sid in batch_ids:\n        bp = bytes_idx.get(sid)\n        ap = asm_idx.get(sid)\n\n        # Image\n        img = ImageOps.process_image(bp, ap, cfg, exact_entropy=False)\n        np.save(img_dir / f'{sid}.npy', img)\n\n        # Phase A: ASM features\n        if ap:\n            counts = OpcodeOps.parse_extended_v2(ap)\n        else:\n            counts = {\n                'uni':{}, 'bi':{}, 'sections':{}, 'imports':[], 'api_counts':{},\n                'funcs':{}, 'strings':{}, 'regs':{}, 'total_lines':0, 'n_reg_lines':0\n            }\n\n        # Phase B: byte stats\n        counts['byte_stats'] = (\n            ImageOps.build_byte_stats(bp) if bp else ImageOps._empty_byte_stats()\n        )\n        counts['id'] = sid\n        raw_records.append(counts)\n\n        FileOps.delete_raw(bp, ap)\n\n    shutil.rmtree(SCRATCH, ignore_errors=True)\n    SCRATCH.mkdir(exist_ok=True)\n\n# Sort theo archive order\nlog.info('Sorting test IDs by archive order...')\ntest_archive_order = FileOps.get_archive_order(COMP_DIR / 'test.7z')\nTEST_IDS_SORTED    = FileOps.sort_by_archive_order(TEST_IDS, test_archive_order)\ntest_batches       = list(FileOps.chunked(TEST_IDS_SORTED, cfg.batch_n))\n\ntest_raw_records = []\nfor batch_ids in tqdm(test_batches, desc='Test batches', ncols=80):\n    process_test_batch(batch_ids, COMP_DIR / 'test.7z',\n                       TEST_IMG_DIR, test_raw_records)\n    gc.collect()\n    GPU.free()\n\n# Vectorize opcodes\ntest_ids_vec, X_test_sparse = OpcodeOps.vectorize(\n    test_raw_records, top_uni, top_bi\n)\ndf_test_tab = pd.DataFrame(\n    X_test_sparse.toarray(),\n    index=test_ids_vec,\n    columns=top_uni + top_bi\n)\ndf_test_tab.index.name = 'Id'\ndf_test_tab.to_csv(TEST_TAB_PATH)\n\n# Phase A extra\ndf_test_extra = OpcodeOps.build_extra_features_v2(test_raw_records)\ndf_test_extra.to_csv(TEST_EXTRA_PATH)\n\n# Phase B byte stats\nbyte_rows = []\nfor r in test_raw_records:\n    row = {'Id': r['id']}\n    row.update(r.get('byte_stats', ImageOps._empty_byte_stats()))\n    byte_rows.append(row)\ndf_test_byte = pd.DataFrame(byte_rows).set_index('Id')\ndf_test_byte.to_csv(TEST_BYTE_PATH)\n\nlog.info('Test features | opcode=%s extra=%s byte=%s',\n         df_test_tab.shape, df_test_extra.shape, df_test_byte.shape)\n\n# Metadata\ntest_bytes_sizes, test_asm_sizes = get_file_sizes_by_ext(COMP_DIR / 'test.7z')\nmeta_rows = []\nfor sid in TEST_IDS:\n    bs = test_bytes_sizes.get(sid, 0)\n    asz = test_asm_sizes.get(sid, 0)\n    meta_rows.append({\n        'Id': sid, 'bytes_size': bs,\n        'asm_size': asz, 'size_ratio': asz/bs if bs > 0 else 0.0,\n    })\ndf_test_meta = pd.DataFrame(meta_rows).set_index('Id')\ndf_test_meta['unique_opcode_count'] = (df_test_tab > 0).sum(axis=1)\ndf_test_meta.to_csv(TEST_META_PATH)\n\n# CNN\nmodel = CNNExtractor().to(DEVICE).eval()\ntest_dataset = MalwareImageDataset(TEST_IDS, TEST_IMG_DIR)\ntest_loader  = DataLoader(test_dataset, batch_size=64, shuffle=False,\n                          num_workers=4, pin_memory=True)\ntest_feats, test_ids_cnn = [], []\nwith torch.no_grad():\n    for imgs, ids in tqdm(test_loader, desc='CNN test', ncols=80):\n        test_feats.append(model(imgs.to(DEVICE)).cpu().numpy())\n        test_ids_cnn.extend(ids)\ncnn_feat_test = np.vstack(test_feats)\nnp.save(TEST_CNN_PATH, cnn_feat_test)\n\nassert cnn_feat_test.std() > 0.1, f\"CNN zeros! std={cnn_feat_test.std():.4f}\"\nlog.info('CNN sanity: std=%.3f zeros=%.1f%%',\n         cnn_feat_test.std(), (cnn_feat_test == 0).mean() * 100)\n\ndel model\ntorch.cuda.empty_cache()\n\n# Load all\ncnn_feat_test = np.load(TEST_CNN_PATH)\ndf_test_tab   = pd.read_csv(TEST_TAB_PATH,   index_col='Id')\ndf_test_extra = pd.read_csv(TEST_EXTRA_PATH, index_col='Id')\ndf_test_byte  = pd.read_csv(TEST_BYTE_PATH,  index_col='Id')\ndf_test_meta  = pd.read_csv(TEST_META_PATH,  index_col='Id')\nlog.info('Test loaded | CNN=%s tab=%s extra=%s byte=%s meta=%s',\n         cnn_feat_test.shape, df_test_tab.shape,\n         df_test_extra.shape, df_test_byte.shape, df_test_meta.shape)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-08T14:21:40.435919Z","iopub.execute_input":"2026-04-08T14:21:40.436199Z","iopub.status.idle":"2026-04-08T16:46:26.372776Z","shell.execute_reply.started":"2026-04-08T14:21:40.436177Z","shell.execute_reply":"2026-04-08T16:46:26.372048Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"##  Assemble full feature matrix","metadata":{}},{"cell_type":"code","source":"# ── Align theo thứ tự ALL_IDS ─────────────────────────────────────────────────\ncnn_id_order = {sid: i for i, sid in enumerate(cnn_ids_train)}\ncnn_idx      = [cnn_id_order[sid] for sid in ALL_IDS]\nX_cnn_train  = cnn_feat_train[cnn_idx]                          # (N, 1280)\n\nX_tab_train   = df_tab.reindex(ALL_IDS).fillna(0).values        # (N, 700)\nX_extra_train = df_extra.reindex(ALL_IDS).fillna(0).values      # (N, ~103) Phase A\nX_byte_train  = df_byte.reindex(ALL_IDS).fillna(0).values       # (N, ~44)  Phase B\n\ndf_meta['unique_opcode_count'] = (df_tab > 0).sum(axis=1)\nX_meta_train = df_meta.reindex(ALL_IDS).fillna(0).values        # (N, 4)\n\n# Test sides\nX_test_cnn   = cnn_feat_test                                              # (M, 1280)\nX_test_tab   = df_test_tab.reindex(TEST_IDS).fillna(0).values             # (M, 700)\nX_test_extra = df_test_extra.reindex(TEST_IDS).fillna(0).values           # (M, ~103)\nX_test_byte  = df_test_byte.reindex(TEST_IDS).fillna(0).values            # (M, ~44)\nX_test_meta  = df_test_meta.reindex(TEST_IDS).fillna(0).values            # (M, 4)\n\ny_train = df_labels.set_index('Id').reindex(ALL_IDS)['label'].values\n\n# Concat — CNN | Opcode | Extra (Phase A) | Byte (Phase B) | Meta\nX_train = np.hstack([X_cnn_train, X_tab_train, X_extra_train, X_byte_train, X_meta_train])\nX_test  = np.hstack([X_test_cnn,  X_test_tab,  X_test_extra,  X_test_byte,  X_test_meta])\n\nassert X_train.shape[1] == X_test.shape[1], \\\n    f'Shape mismatch: train={X_train.shape[1]} test={X_test.shape[1]}'\n\nlog.info('X_train shape: %s (CNN=%d + Tab=%d + Extra=%d + Byte=%d + Meta=%d)',\n         X_train.shape,\n         X_cnn_train.shape[1],   X_tab_train.shape[1],\n         X_extra_train.shape[1], X_byte_train.shape[1],\n         X_meta_train.shape[1])\nlog.info('X_test shape : %s', X_test.shape)\n\nprint(f'\\nX_train : {X_train.shape}')\nprint(f'X_test  : {X_test.shape}')\nprint(f'y_train : {y_train.shape}  classes={np.unique(y_train)}')\nprint(f'\\nClass distribution:')\nfor c, n in sorted(zip(*np.unique(y_train, return_counts=True))):\n    print(f'  Class {c+1} ({CLASS_NAMES[c+1]:<18}): {n:4d} ({n/len(y_train)*100:.1f}%)')\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-08T16:46:26.37456Z","iopub.execute_input":"2026-04-08T16:46:26.374871Z","iopub.status.idle":"2026-04-08T16:46:27.03585Z","shell.execute_reply.started":"2026-04-08T16:46:26.374787Z","shell.execute_reply":"2026-04-08T16:46:27.035223Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Cell 7 — Feature Selection (Random Forest, Outside CV)","metadata":{}},{"cell_type":"code","source":"TOP_K = 1000\nSEL_PATH = FEAT_DIR / 'selected_indices.npy'\n\n# Force recompute — feature space changed (2007 → ~2150)\nif SEL_PATH.exists():\n    SEL_PATH.unlink()\n    log.info('Deleted stale selected_indices.npy (feature space changed)')\n\nlog.info('Fitting Random Forest for feature importance (n_estimators=200)...')\nt0 = time.time()\n\nrf = RandomForestClassifier(\n    n_estimators=200, max_depth=15,\n    n_jobs=-1, random_state=SEED\n)\nrf.fit(X_train, y_train)\n\nimportances  = rf.feature_importances_\nselected_idx = np.argsort(importances)[::-1][:TOP_K]\nnp.save(SEL_PATH, selected_idx)\n\nlog.info('Feature selection done in %.1f min', (time.time()-t0)/60)\n\nX_train_sel = X_train[:, selected_idx]\nX_test_sel  = X_test[:, selected_idx]\n\nlog.info('After selection: train=%s  test=%s', X_train_sel.shape, X_test_sel.shape)\n\n# Breakdown\nCNN_N   = X_cnn_train.shape[1]\nTAB_N   = X_tab_train.shape[1]\nEXTRA_N = X_extra_train.shape[1]\nBYTE_N  = X_byte_train.shape[1]\nMETA_N  = X_meta_train.shape[1]\n\nb1 = CNN_N\nb2 = b1 + TAB_N\nb3 = b2 + EXTRA_N\nb4 = b3 + BYTE_N\nb5 = b4 + META_N\n\nn_cnn_sel   = int((selected_idx < b1).sum())\nn_tab_sel   = int(((selected_idx >= b1) & (selected_idx < b2)).sum())\nn_extra_sel = int(((selected_idx >= b2) & (selected_idx < b3)).sum())\nn_byte_sel  = int(((selected_idx >= b3) & (selected_idx < b4)).sum())\nn_meta_sel  = int(((selected_idx >= b4) & (selected_idx < b5)).sum())\n\nprint(f'\\nTop-{TOP_K} feature selection breakdown:')\nprint(f'  CNN   : {n_cnn_sel:4d}/{CNN_N}')\nprint(f'  Opcode: {n_tab_sel:4d}/{TAB_N}')\nprint(f'  Extra : {n_extra_sel:4d}/{EXTRA_N}  (Phase A: sec/api/func/string/reg)')\nprint(f'  Byte  : {n_byte_sel:4d}/{BYTE_N}   (Phase B: byte stats)')\nprint(f'  Meta  : {n_meta_sel:4d}/{META_N}')\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-08T16:46:27.037058Z","iopub.execute_input":"2026-04-08T16:46:27.03739Z","iopub.status.idle":"2026-04-08T16:46:49.964626Z","shell.execute_reply.started":"2026-04-08T16:46:27.037366Z","shell.execute_reply":"2026-04-08T16:46:49.964004Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"##  Stratified K-Fold XGBoost Training","metadata":{}},{"cell_type":"code","source":"# ── XGBoost params ────────────────────────────────────────────────────────────\nXGB_PARAMS = {\n    'objective'        : 'multi:softprob',\n    'num_class'        : N_CLASSES,\n    'eval_metric'      : 'mlogloss',\n    'max_depth'        : 7,\n    'learning_rate'    : 0.05,\n    'subsample'        : 0.8,\n    'colsample_bytree' : 0.8,\n    'min_child_weight' : 3,\n    'gamma'            : 0.1,\n    'tree_method'      : 'hist',   # GPU acceleration\n    'device'           : 'cuda',\n    'seed'             : SEED,\n    'verbosity'        : 0,\n}\n\n# ── sample_weight cho class imbalance ─────────────────────────────────────────\n# Đúng cách cho multiclass — scale_pos_weight bị ignore trong multi:softprob\nsample_weights = compute_sample_weight('balanced', y_train)\nlog.info('Sample weight range: [%.3f, %.3f]',\n         sample_weights.min(), sample_weights.max())\n\n# ── Stratified K-Fold ─────────────────────────────────────────────────────────\nskf = StratifiedKFold(n_splits=N_FOLDS, shuffle=True, random_state=SEED)\n\noof_preds   = np.zeros((len(y_train), N_CLASSES))  # Out-of-fold predictions\ntest_preds  = np.zeros((len(TEST_IDS), N_CLASSES)) # Test predictions\nfold_models = []\nfold_scores = []\n\nlog.info('Starting %d-Fold XGBoost training...', N_FOLDS)\nt_total = time.time()\n\nfor fold, (tr_idx, val_idx) in enumerate(\n        skf.split(X_train_sel, y_train)):\n\n    t_fold = time.time()\n    log.info('Fold %d/%d ...', fold+1, N_FOLDS)\n\n    X_tr, X_val = X_train_sel[tr_idx], X_train_sel[val_idx]\n    y_tr, y_val = y_train[tr_idx],     y_train[val_idx]\n    w_tr        = sample_weights[tr_idx]\n\n    dtrain = xgb.DMatrix(X_tr,  label=y_tr,  weight=w_tr)\n    dval   = xgb.DMatrix(X_val, label=y_val)\n    dtest  = xgb.DMatrix(X_test_sel)\n\n    model = xgb.train(\n        XGB_PARAMS,\n        dtrain,\n        num_boost_round=1000,\n        evals=[(dval, 'val')],\n        early_stopping_rounds=30,\n        verbose_eval=100\n    )\n\n    # OOF predictions\n    val_prob = model.predict(dval)                # (val_size, 9)\n    oof_preds[val_idx] = val_prob\n\n    # Test predictions (accumulate)\n    test_preds += model.predict(dtest) / N_FOLDS\n\n    # Score\n    val_acc  = accuracy_score(y_val, val_prob.argmax(axis=1))\n    val_loss = log_loss(y_val, val_prob)\n    fold_scores.append({'acc': val_acc, 'loss': val_loss})\n    fold_models.append(model)\n\n    log.info('Fold %d | acc=%.4f | log_loss=%.4f | %.1fs',\n             fold+1, val_acc, val_loss, time.time()-t_fold)\n\n# ── OOF overall score ─────────────────────────────────────────────────────────\noof_acc  = accuracy_score(y_train, oof_preds.argmax(axis=1))\noof_loss = log_loss(y_train, oof_preds)\nlog.info('OOF overall | acc=%.4f | log_loss=%.4f | total=%.1f min',\n         oof_acc, oof_loss, (time.time()-t_total)/60)\n\nprint(f'\\n{\"─\"*40}')\nprint(f' OOF Accuracy : {oof_acc:.4f}')\nprint(f' OOF Log-loss : {oof_loss:.4f}')\nprint(f'{'─'*40}')\nfor i, s in enumerate(fold_scores):\n    print(f' Fold {i+1}: acc={s[\"acc\"]:.4f}  loss={s[\"loss\"]:.4f}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-08T16:46:49.96588Z","iopub.execute_input":"2026-04-08T16:46:49.966206Z","iopub.status.idle":"2026-04-08T16:48:14.68119Z","shell.execute_reply.started":"2026-04-08T16:46:49.96618Z","shell.execute_reply":"2026-04-08T16:48:14.68051Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"##  Pseudo-labeling ","metadata":{}},{"cell_type":"code","source":"PSEUDO_THRESHOLD = 0.98   # chỉ lấy predictions confidence > 98%\nMAX_PSEUDO_ROUNDS = 2\n\nX_aug  = X_train_sel.copy()\ny_aug  = y_train.copy()\nw_aug  = sample_weights.copy()\n\nfor pseudo_round in range(MAX_PSEUDO_ROUNDS):\n    # Tìm high-confidence test predictions\n    max_probs   = test_preds.max(axis=1)          # probability của class tự tin nhất\n    pseudo_mask = max_probs >= PSEUDO_THRESHOLD\n    pseudo_y    = test_preds.argmax(axis=1)\n\n    n_pseudo = pseudo_mask.sum()\n    log.info('Pseudo round %d: %d/%d samples (threshold=%.2f)',\n             pseudo_round+1, n_pseudo, len(TEST_IDS), PSEUDO_THRESHOLD)\n\n    if n_pseudo == 0:\n        log.info('Không có pseudo-labels — dừng')\n        break\n\n    # Thêm pseudo-labeled test vào training\n    X_pseudo = X_test_sel[pseudo_mask]\n    y_pseudo = pseudo_y[pseudo_mask]\n    w_pseudo = compute_sample_weight('balanced',\n                                     np.concatenate([y_aug, y_pseudo]))[len(y_aug):]\n\n    X_aug = np.vstack([X_aug, X_pseudo])\n    y_aug = np.concatenate([y_aug, y_pseudo])\n    w_aug = np.concatenate([w_aug, w_pseudo])\n\n    log.info('Augmented train size: %d', len(y_aug))\n\n    # Retrain trên toàn bộ augmented data\n    dtrain_aug = xgb.DMatrix(X_aug,      label=y_aug, weight=w_aug)\n    dtest_xgb  = xgb.DMatrix(X_test_sel)\n\n    # Số rounds = best round từ fold training trung bình\n    best_rounds = int(np.mean([m.best_iteration for m in fold_models]))\n\n    pseudo_model = xgb.train(\n        {**XGB_PARAMS, 'verbosity': 0},\n        dtrain_aug,\n        num_boost_round=best_rounds,\n    )\n\n    # Update test predictions\n    new_test_preds = pseudo_model.predict(dtest_xgb)\n\n    # Kiểm tra improvement trên OOF (dùng làm proxy)\n    doof = xgb.DMatrix(X_train_sel)\n    new_oof = pseudo_model.predict(doof)\n    new_oof_acc  = accuracy_score(y_train, new_oof.argmax(axis=1))\n    new_oof_loss = log_loss(y_train, new_oof)\n    log.info('After pseudo round %d: OOF acc=%.4f (was %.4f) | loss=%.4f (was %.4f)',\n             pseudo_round+1, new_oof_acc, oof_acc, new_oof_loss, oof_loss)\n\n    if new_oof_acc > oof_acc:\n        test_preds = new_test_preds\n        oof_acc    = new_oof_acc\n        log.info('✅ Pseudo-labeling cải thiện — giữ model mới')\n    else:\n        log.info('⚠  Pseudo-labeling không cải thiện — giữ nguyên')\n        break","metadata":{"execution":{"iopub.status.busy":"2026-04-08T16:48:14.682257Z","iopub.execute_input":"2026-04-08T16:48:14.682534Z","iopub.status.idle":"2026-04-08T16:49:11.833206Z","shell.execute_reply.started":"2026-04-08T16:48:14.682513Z","shell.execute_reply":"2026-04-08T16:49:11.832598Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ── Ensemble: Soft voting ─────────────────────────────────────────────────────\n# test_preds đã là average của 5 folds (Cell 8) + pseudo-labeling update\n# Final class = argmax(probabilities)\nfinal_class = test_preds.argmax(axis=1) + 1   # +1 vì label 0-indexed\n\n# ── OOF Analysis ─────────────────────────────────────────────────────────────\nprint('OOF CONFUSION MATRIX')\noof_class = oof_preds.argmax(axis=1)\n\nfrom sklearn.metrics import classification_report, confusion_matrix\n\nprint(classification_report(\n    y_train, oof_class,\n    target_names=[CLASS_NAMES[i+1] for i in range(9)]\n))\n\n# Per-class accuracy\nprint('\\nPer-class accuracy:')\nfor c in range(9):\n    mask = y_train == c\n    if mask.sum() > 0:\n        acc = (oof_class[mask] == c).mean()\n        n   = mask.sum()\n        flag = '⚠' if acc < 0.95 else '✅'\n        print(f'  {flag} Class {c+1} ({CLASS_NAMES[c+1]:<18}): '\n              f'{acc:.4f}  (n={n})')\n\n# ── Create submission ─────────────────────────────────────────────────────────\n# ── Create submission ─────────────────────────────────────────────────────────\n\ncolumn_names = [f'Prediction{i}' for i in range(1, 10)]\ndf_submission = pd.DataFrame(test_preds, columns=column_names)\n\ndf_submission.insert(0, 'Id', TEST_IDS)\n\nsub_path = WORK_DIR / 'submission.csv'\ndf_submission.to_csv(sub_path, index=False)\n\nprint(f'\\nSubmission saved: {sub_path}')\nprint(f'Shape: {df_submission.shape}') # Sẽ in ra (10873, 10)\ndisplay(df_submission.head())\n\n# ── Final summary ─────────────────────────────────────────────────────────────\nprint(f'\\n{\"═\"*45}')\nprint(f'  FINAL RESULTS')\nprint(f'  OOF Accuracy  : {oof_acc:.4f}')\nprint(f'  OOF Log-loss  : {oof_loss:.4f}')\nprint(f'  Target        : ≥ 0.99')\nstatus = '✅ ĐẠT' if oof_acc >= 0.99 else '⚠  CHƯA ĐẠT'\nprint(f'  Status        : {status}')\nprint(f'{'═'*45}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-09T04:45:43.503739Z","iopub.execute_input":"2026-04-09T04:45:43.503973Z","iopub.status.idle":"2026-04-09T04:45:43.523821Z","shell.execute_reply.started":"2026-04-09T04:45:43.503946Z","shell.execute_reply":"2026-04-09T04:45:43.522997Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Ensemble + Submission","metadata":{}}]}