{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"codemirror_mode":{"name":"ipython","version":3},"file_extension":".py","mimetype":"text/x-python","name":"python","nbconvert_exporter":"python","pygments_lexer":"ipython3","version":"3.12.12"},"kaggle":{"accelerator":"none","dataSources":[{"databundleVersionId":46665,"sourceId":4117,"sourceType":"competition"}],"dockerImageVersionId":31328,"isGpuEnabled":false,"isInternetEnabled":true,"language":"python","sourceType":"notebook"},"papermill":{"default_parameters":{},"duration":30934.424603,"end_time":"2026-04-13T17:50:53.866309+00:00","environment_variables":{},"exception":null,"input_path":"__notebook__.ipynb","output_path":"__notebook__.ipynb","parameters":{},"start_time":"2026-04-13T09:15:19.441706+00:00","version":"2.7.0"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Improved Data Processing — Malware Classification\n\n## Hướng dẫn sử dụng\nNotebook này dùng chung cho **12 lần chạy song song**:\n- **4 notebooks TRAIN**: `TARGET=\"TRAIN\"`, `TOTAL_PARTS=4`, `PART=1/2/3/4`\n- **8 notebooks TEST** : `TARGET=\"TEST\"`,  `TOTAL_PARTS=8`, `PART=1/2/3/4/5/6/7/8`\n\nChỉ cần thay đổi 3 biến ở **Cell 2 (CONTROL SWITCHES)**.\n\n## Features được extract (cải tiến so với phiên bản cũ)\n| Feature group | Số features | Ghi chú |\n|---|---|---|\n| Tabular (meta) | ~75 | Expanded opcodes, registers, entropy x16 windows |\n| Byte N-grams | 100,000 | HashingVec (sparse .npz) — MI selection ở training |\n| Opcode sequences | text corpus | Lưu pkl → CountVec fit ở training notebook |\n| ASM pixel density | 1,024 | 32×32 grid mean byte intensity |\n\n## Output files (lưu vào `/kaggle/working/`)\n```\nX_train_tab_part1.npy        X_test_tab_part1.npy\nX_train_ng_part1.npz         X_test_ng_part1.npz\nX_train_opseq_part1.pkl      X_test_opseq_part1.pkl\nX_train_pixel_part1.npy      X_test_pixel_part1.npy\ny_train_part1.npy            (chỉ có ở TRAIN)\n```","metadata":{"papermill":{"duration":0.003948,"end_time":"2026-04-13T09:15:23.207116+00:00","exception":false,"start_time":"2026-04-13T09:15:23.203168+00:00","status":"completed"},"tags":[]}},{"cell_type":"code","source":"# ============================================================\n# CELL 1 — IMPORTS & THƯ VIỆN\n# ============================================================\nimport os\nimport re\nimport gc\nimport math\nimport zlib\nimport pickle\nimport logging\nimport subprocess\nimport numpy as np\nimport pandas as pd\nimport scipy.sparse as sp\nfrom tqdm.notebook import tqdm\nfrom multiprocessing import Pool\nfrom scipy.stats import entropy\nfrom sklearn.feature_extraction.text import HashingVectorizer\n\nprint(\"✅ Import xong!\")","metadata":{"execution":{"iopub.execute_input":"2026-04-13T09:15:23.214799Z","iopub.status.busy":"2026-04-13T09:15:23.214409Z","iopub.status.idle":"2026-04-13T09:15:26.818537Z","shell.execute_reply":"2026-04-13T09:15:26.817153Z"},"papermill":{"duration":3.610445,"end_time":"2026-04-13T09:15:26.820701+00:00","exception":false,"start_time":"2026-04-13T09:15:23.210256+00:00","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ============================================================\n# CELL 2 — CONTROL SWITCHES (CHỈ THAY ĐỔI Ở ĐÂY)\n# ============================================================\n\nTARGET      = \"TEST\"  # ← \"TRAIN\" hoặc \"TEST\"\nTOTAL_PARTS = 8        # ← 4 nếu TRAIN, 8 nếu TEST\nPART        = 8        # ← Số thứ tự notebook này (1 đến TOTAL_PARTS)\n\n# ============================================================\n# CẤU HÌNH ĐƯỜNG DẪN (không cần sửa)\n# ============================================================\nDATA_DIR       = '/kaggle/input/competitions/malware-classification'\nTRAIN_7Z       = os.path.join(DATA_DIR, 'train.7z')\nTEST_7Z        = os.path.join(DATA_DIR, 'test.7z')\nLABEL_CSV      = os.path.join(DATA_DIR, 'trainLabels.csv')\nSUBMISSION_CSV = os.path.join(DATA_DIR, 'sampleSubmission.csv')\n\nWORKING_DIR = '/kaggle/working'\nTEMP_DIR    = os.path.join(WORKING_DIR, 'temp_unzip')\n\nos.makedirs(TEMP_DIR, exist_ok=True)\n\n# Số file giải nén mỗi lượt (tăng nếu RAM > 13GB, giảm nếu thiếu disk)\nBATCH_SIZE = 50\n\nprint(f\"🎯 Cấu hình: TARGET={TARGET}, PART={PART}/{TOTAL_PARTS}\")\nprint(f\"📁 Kết quả lưu tại: {WORKING_DIR}\")","metadata":{"execution":{"iopub.execute_input":"2026-04-13T09:15:26.828779Z","iopub.status.busy":"2026-04-13T09:15:26.828274Z","iopub.status.idle":"2026-04-13T09:15:26.836341Z","shell.execute_reply":"2026-04-13T09:15:26.8354Z"},"papermill":{"duration":0.014668,"end_time":"2026-04-13T09:15:26.83858+00:00","exception":false,"start_time":"2026-04-13T09:15:26.823912+00:00","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ============================================================\n# CELL 3 — LOGGING & CẤU HÌNH VECTORIZER\n# ============================================================\n\n# Setup logging ra cả file và console\nlog_file = os.path.join(WORKING_DIR, f'processing_{TARGET}_part{PART}.log')\nlogging.basicConfig(\n    level=logging.INFO,\n    format='%(asctime)s [%(levelname)s] %(message)s',\n    datefmt='%H:%M:%S',\n    handlers=[\n        logging.FileHandler(log_file),\n        logging.StreamHandler()\n    ]\n)\n\n# -------------------------------------------------------------------\n# BYTE N-GRAM VECTORIZER\n# Dùng HashingVectorizer với n_features=100000 để giảm thiểu\n# hash collision (so với 20000 cũ). Không cần fit → có thể dùng\n# trực tiếp mà không cần thấy toàn bộ dữ liệu trước.\n# Feature selection bằng mutual_info sẽ thực hiện ở training notebook.\n# -------------------------------------------------------------------\nBYTE_VECTORIZER = HashingVectorizer(\n    n_features=2**17,       # 131072 ≈ 130K buckets — giảm collision đáng kể\n    ngram_range=(2, 4),     # bigram đến 4-gram\n    lowercase=True,\n    norm=None,              # raw counts, không normalize — MI selection cần raw\n    alternate_sign=False    # đảm bảo giá trị không âm cho MI\n)\n\nprint(\"✅ Logger và vectorizer đã sẵn sàng\")","metadata":{"execution":{"iopub.execute_input":"2026-04-13T09:15:26.846509Z","iopub.status.busy":"2026-04-13T09:15:26.846177Z","iopub.status.idle":"2026-04-13T09:15:26.853818Z","shell.execute_reply":"2026-04-13T09:15:26.852837Z"},"papermill":{"duration":0.013955,"end_time":"2026-04-13T09:15:26.855733+00:00","exception":false,"start_time":"2026-04-13T09:15:26.841778+00:00","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ============================================================\n# CELL 4 — ĐỊNH NGHĨA FEATURE SETS (MỞ RỘNG)\n# ============================================================\n\n# --- Opcodes: 35 instructions bao gồm cả jmp variants ---\n# Cũ: 12 opcodes. Mới: 35 opcodes bao gồm control-flow đầy đủ.\nOPCODES = [\n    # Arithmetic & logic\n    'add', 'sub', 'mul', 'div', 'imul', 'idiv', 'inc', 'dec',\n    'and', 'or', 'xor', 'not', 'neg', 'shl', 'shr', 'sar', 'rol', 'ror',\n    # Data transfer\n    'mov', 'movsx', 'movzx', 'lea', 'push', 'pop', 'xchg',\n    # Control flow — phân biệt jmp variants rất quan trọng!\n    'call', 'ret', 'jmp',\n    'je', 'jne', 'jz', 'jnz', 'jg', 'jge', 'jl', 'jle', 'ja', 'jb', 'loop',\n    # Compare & misc\n    'cmp', 'test', 'nop', 'int', 'hlt',\n]\n\n# --- Registers: 16 registers ---\n# Feature mới: đếm tần suất sử dụng registers → phản ánh kiến trúc malware\nREGISTERS = [\n    'eax', 'ebx', 'ecx', 'edx', 'esi', 'edi', 'esp', 'ebp',  # 32-bit\n    'ax', 'bx', 'cx', 'dx',                                    # 16-bit\n    'al', 'bl', 'cl', 'dl',                                    # 8-bit low\n]\n\n# --- Sections: mở rộng thêm sections thường gặp trong malware ---\nSECTIONS = [\n    'HEADER', '.text', '.data', '.rdata', '.bss',\n    '.idata', '.edata', '.rsrc', '.reloc', 'UPX0', 'UPX1',\n]\n\n# --- Opcodes hợp lệ cho việc extract sequence (lowercase set) ---\n# Dùng set để O(1) lookup khi quét từng dòng\nVALID_OPCODE_SET = set(OPCODES)\n\nprint(f\"📊 Feature sets:\")\nprint(f\"  Opcodes: {len(OPCODES)} (cũ: 12, mới: {len(OPCODES)})\")\nprint(f\"  Registers: {len(REGISTERS)} (cũ: 0, mới: {len(REGISTERS)})\")\nprint(f\"  Sections: {len(SECTIONS)} (cũ: 7, mới: {len(SECTIONS)})\")","metadata":{"execution":{"iopub.execute_input":"2026-04-13T09:15:26.864238Z","iopub.status.busy":"2026-04-13T09:15:26.863566Z","iopub.status.idle":"2026-04-13T09:15:26.872574Z","shell.execute_reply":"2026-04-13T09:15:26.871421Z"},"papermill":{"duration":0.01539,"end_time":"2026-04-13T09:15:26.874413+00:00","exception":false,"start_time":"2026-04-13T09:15:26.859023+00:00","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ============================================================\n# CELL 5 — HÀM TRÍCH XUẤT TABULAR FEATURES (NÂNG CẤP)\n# ============================================================\n\ndef calculate_entropy(data):\n    \"\"\"Tính Shannon entropy cho mảng byte values.\"\"\"\n    if len(data) == 0:\n        return 0.0\n    p, _ = np.histogram(data, bins=256, range=(0, 255), density=True)\n    return float(entropy(p + 1e-10))\n\n\ndef extract_entropy_windows(values, n_windows=16):\n    \"\"\"\n    Chia chuỗi byte thành n_windows đoạn, tính entropy từng đoạn.\n    Cũ: 4 quartile. Mới: 16 windows → bắt được pattern biến thiên entropy\n    phức tạp hơn (vùng packed, vùng plain text, vùng resources...).\n    \"\"\"\n    if len(values) == 0:\n        return [0.0] * n_windows\n    chunks = np.array_split(values, n_windows)\n    return [calculate_entropy(c) for c in chunks]\n\n\ndef extract_file_properties(file_path):\n    \"\"\"Đọc file binary, trả về (raw_size, compressed_size).\"\"\"\n    try:\n        with open(file_path, 'rb') as f:\n            data = f.read()\n        return len(data), len(zlib.compress(data, level=1))\n    except Exception:\n        return 0, 0\n\n\ndef process_single_sample(sample_id, folder_path):\n    \"\"\"\n    Trích xuất TOÀN BỘ tabular features cho 1 mẫu malware.\n    \n    Returns:\n        dict: Tất cả features có key, sẵn sàng cho pd.DataFrame\n    \"\"\"\n    asm_path   = os.path.join(folder_path, f\"{sample_id}.asm\")\n    bytes_path = os.path.join(folder_path, f\"{sample_id}.bytes\")\n    features   = {'Id': sample_id}\n\n    # ── 1. Kích thước & tỷ lệ nén ──────────────────────────────────────\n    bytes_raw,  bytes_comp = extract_file_properties(bytes_path)\n    asm_raw,    asm_comp   = extract_file_properties(asm_path)\n\n    features['bytes_raw_size'] = bytes_raw\n    features['asm_raw_size']   = asm_raw\n    features['ab_ratio']       = asm_raw  / (bytes_raw + 1)   # tỷ lệ asm/bytes\n    features['bytes_comp_ratio'] = bytes_comp / (bytes_raw + 1)\n    features['asm_comp_ratio']   = asm_comp   / (asm_raw + 1)\n    # abc_ratio: độ nén tương đối asm so với bytes → detect packing\n    features['abc_ratio'] = features['asm_comp_ratio'] / (features['bytes_comp_ratio'] + 1e-5)\n\n    # ── 2. Khởi tạo counters ────────────────────────────────────────────\n    section_counts  = {s:  0 for s in SECTIONS}\n    opcode_counts   = {op: 0 for op in OPCODES}\n    register_counts = {r:  0 for r  in REGISTERS}\n\n    total_lines       = 0\n    total_opcode_lines = 0   # dòng có opcode hợp lệ → tính control-flow ratio\n    jmp_total         = 0    # tổng tất cả jmp variants\n    call_total        = 0\n\n    # ── 3. Quét file .asm ───────────────────────────────────────────────\n    if os.path.exists(asm_path):\n        with open(asm_path, 'r', encoding='latin-1', errors='ignore') as f:\n            for line in f:\n                total_lines += 1\n                line_lower  = line.lower().strip()\n                tokens      = line_lower.split()\n\n                # Section headers\n                for sec in SECTIONS:\n                    if line_lower.startswith(sec.lower()):\n                        section_counts[sec] += 1\n\n                # Opcodes: quét qua các token để tìm opcode hợp lệ (bỏ qua địa chỉ và hex bytes)\n                if tokens:\n                    import re\n                    for tok in tokens:\n                        clean_tok = tok.rstrip(':;,')\n                        # Bỏ qua các chuỗi hex 2 ký tự (ví dụ: 55, 8b)\n                        if len(clean_tok) == 2 and re.match(r'[0-9a-f]{2}$', clean_tok):\n                            continue\n                        # Bỏ qua prefix địa chỉ (ví dụ: .text:00401000)\n                        if ':' in tok:\n                            continue\n                            \n                        # Nếu tìm thấy opcode\n                        if clean_tok in VALID_OPCODE_SET:\n                            opcode_counts[clean_tok] += 1\n                            total_opcode_lines += 1\n                            if clean_tok in ('jmp','je','jne','jz','jnz','jg','jge','jl','jle','ja','jb','loop'):\n                                jmp_total += 1\n                            if clean_tok == 'call':\n                                call_total += 1\n                            break  # Mỗi dòng chỉ tính 1 opcode chính\n\n                # Registers: scan toàn dòng\n                for reg in REGISTERS:\n                    # Dùng word boundary để không match \"eax\" trong \"heaxdump\"\n                    if f' {reg}' in line_lower or f',{reg}' in line_lower:\n                        register_counts[reg] += 1\n\n    features['total_asm_lines']      = total_lines\n    features['total_opcode_lines']   = total_opcode_lines\n    # Control-flow ratio: tỷ lệ jmp và call trong tổng opcode → detect obfuscation\n    features['jmp_ratio']  = jmp_total  / (total_opcode_lines + 1)\n    features['call_ratio'] = call_total / (total_opcode_lines + 1)\n    features['jmp_to_call_ratio'] = jmp_total / (call_total + 1)\n\n    # Lưu section, opcode, register counts\n    features.update({f'sec_{k}':  v for k, v in section_counts.items()})\n    features.update({f'op_{k}':   v for k, v in opcode_counts.items()})\n    features.update({f'reg_{k}':  v for k, v in register_counts.items()})\n\n    # ── 4. Entropy features từ file .bytes (16 windows) ─────────────────\n    if os.path.exists(bytes_path):\n        with open(bytes_path, 'r', errors='ignore') as f:\n            hex_tokens  = f.read().split()\n            valid_bytes = [int(h, 16) for h in hex_tokens if len(h) == 2 and h != '??']\n\n        if valid_bytes:\n            arr = np.array(valid_bytes, dtype=np.uint8)\n            features['entropy_global']   = calculate_entropy(arr)\n            features['byte_mean']        = float(arr.mean())\n            features['byte_std']         = float(arr.std())\n            features['byte_unique_ratio']= len(np.unique(arr)) / 256.0\n            features['null_byte_ratio']  = float((arr == 0).mean())\n            features['high_byte_ratio']  = float((arr > 127).mean())\n\n            # 16 entropy windows\n            window_entropies = extract_entropy_windows(arr, n_windows=16)\n            for i, e in enumerate(window_entropies):\n                features[f'entropy_w{i:02d}'] = e\n\n            # Entropy statistics across windows → detect local packing\n            features['entropy_win_std']  = float(np.std(window_entropies))\n            features['entropy_win_max']  = float(np.max(window_entropies))\n            features['entropy_win_min']  = float(np.min(window_entropies))\n        else:\n            features['entropy_global']   = 0.0\n            features['byte_mean']        = 0.0\n            features['byte_std']         = 0.0\n            features['byte_unique_ratio']= 0.0\n            features['null_byte_ratio']  = 0.0\n            features['high_byte_ratio']  = 0.0\n            for i in range(16):\n                features[f'entropy_w{i:02d}'] = 0.0\n            features['entropy_win_std']  = 0.0\n            features['entropy_win_max']  = 0.0\n            features['entropy_win_min']  = 0.0\n    else:\n        # File không tồn tại: điền zero cho tất cả entropy features\n        for key in ['entropy_global','byte_mean','byte_std','byte_unique_ratio',\n                    'null_byte_ratio','high_byte_ratio','entropy_win_std',\n                    'entropy_win_max','entropy_win_min']:\n            features[key] = 0.0\n        for i in range(16):\n            features[f'entropy_w{i:02d}'] = 0.0\n\n    return features\n\nprint(f\"✅ Tabular extractor sẵn sàng\")\nprint(f\"   Ước tính số features tabular: ~{6 + len(SECTIONS) + len(OPCODES) + len(REGISTERS) + 5 + 16 + 3 + 5}\")","metadata":{"execution":{"iopub.execute_input":"2026-04-13T09:15:26.883318Z","iopub.status.busy":"2026-04-13T09:15:26.882871Z","iopub.status.idle":"2026-04-13T09:15:26.908804Z","shell.execute_reply":"2026-04-13T09:15:26.907815Z"},"papermill":{"duration":0.033168,"end_time":"2026-04-13T09:15:26.911006+00:00","exception":false,"start_time":"2026-04-13T09:15:26.877838+00:00","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ============================================================\n# CELL 6 — HÀM TRÍCH XUẤT OPCODE SEQUENCE (FIXED + TRUNCATED)\n# ============================================================\n# FIX BUG (v2): Format IDA Pro BIG2015: \".text:00401000  55  push  ebp\"\n# → scan tất cả tokens, bỏ hex bytes 2 ký tự, lấy opcode đầu tiên.\n#\n# FIX PERFORMANCE: Truncate tối đa MAX_OPSEQ_TOKENS per file.\n# Lý do: file .asm lớn có 100K-500K opcodes; CountVec/HashVec trên\n# corpus hàng tỷ tokens sẽ chạy nhiều giờ / kill kernel.\n# 10,000 opcodes đầu đã đủ capture entry-point + main logic.\n\nimport re as _re\n\nMAX_OPSEQ_TOKENS = 10_000   # giới hạn tokens/file để tránh timeout\n\ndef extract_opcode_sequence(asm_path, max_tokens=MAX_OPSEQ_TOKENS):\n    \"\"\"\n    Trích xuất chuỗi opcodes từ file .asm, truncate ở max_tokens.\n    Output: string \"push mov sub call xor ...\" để CountVec/HashVec xử lý.\n    \"\"\"\n    opcode_tokens = []\n    if not os.path.exists(asm_path):\n        return \"\"\n\n    with open(asm_path, 'r', encoding='latin-1', errors='ignore') as f:\n        for line in f:\n            if len(opcode_tokens) >= max_tokens:\n                break               # đã đủ, dừng sớm tiết kiệm thời gian\n\n            line_stripped = line.lower().strip()\n            if not line_stripped or line_stripped.startswith(';'):\n                continue\n\n            tokens = line_stripped.split()\n            for token in tokens:\n                clean = token.rstrip(':;,')\n                # Bỏ hex byte 2 ký tự (vd: \"55\", \"8b\")\n                if len(clean) == 2 and _re.match(r'[0-9a-f]{2}$', clean):\n                    continue\n                # Bỏ địa chỉ có dấu \":\" (vd: \".text:00401000\")\n                if ':' in token:\n                    continue\n                if clean in VALID_OPCODE_SET:\n                    opcode_tokens.append(clean)\n                    break           # 1 opcode/dòng\n\n    return ' '.join(opcode_tokens)\n\n\n# ── Unit test nhanh ──────────────────────────────────────────\ndef _test_opcode_extractor():\n    import re as re_\n    test_lines = [\n        \".text:00401000  55                    push    ebp\",\n        \".text:00401001  8B EC                 mov     ebp, esp\",\n        \".text:00401003  83 EC 14              sub     esp, 14h\",\n        \"00401006  E8 F5 FF FF FF        call    sub_401000\",\n        \"   xor     eax, eax\",     # format không có địa chỉ\n        \"; comment line\",           # comment → bỏ qua\n    ]\n    expected = ['push', 'mov', 'sub', 'call', 'xor']\n    result = []\n    for line in test_lines:\n        ls = line.lower().strip()\n        if not ls or ls.startswith(';'): continue\n        for tok in ls.split():\n            c = tok.rstrip(':;,')\n            if len(c) == 2 and re_.match(r'[0-9a-f]{2}$', c): continue\n            if ':' in tok: continue\n            if c in VALID_OPCODE_SET:\n                result.append(c); break\n    assert result == expected, f\"Test failed: {result} != {expected}\"\n    print(f\"  ✅ Unit test passed: {result}\")\n\n_test_opcode_extractor()\nprint(f\"Opcode extractor sẵn sàng (max {MAX_OPSEQ_TOKENS:,} tokens/file)\")","metadata":{"execution":{"iopub.execute_input":"2026-04-13T09:15:26.919889Z","iopub.status.busy":"2026-04-13T09:15:26.919504Z","iopub.status.idle":"2026-04-13T09:15:26.932136Z","shell.execute_reply":"2026-04-13T09:15:26.931005Z"},"papermill":{"duration":0.019485,"end_time":"2026-04-13T09:15:26.933951+00:00","exception":false,"start_time":"2026-04-13T09:15:26.914466+00:00","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ============================================================\n# CELL 7 — HÀM TRÍCH XUẤT ASM PIXEL DENSITY (FEATURE MỚI)\n# ============================================================\n# Phương pháp được đội giải nhất Microsoft BIG2015 gọi là\n# \"most powerful yet simple feature\":\n# 1. Đọc file .asm dưới dạng binary\n# 2. Chia thành GRID_H × GRID_W ô (32×32 = 1024 ô)\n# 3. Tính mean byte value mỗi ô → vector 1024 chiều\n#\n# Tại sao hiệu quả:\n# - Encode CẤU TRÚC KHÔNG GIAN của file asm (các vùng code, data, string...)\n# - Mỗi malware family có visual signature riêng (như fingerprint)\n# - Không cần DL — chỉ là statistical features đơn giản\n\nPIXEL_GRID_H = 32  # Số hàng\nPIXEL_GRID_W = 32  # Số cột → 32×32 = 1024 features\nN_PIXEL_FEATURES = PIXEL_GRID_H * PIXEL_GRID_W\n\ndef extract_pixel_density(asm_path, grid_h=PIXEL_GRID_H, grid_w=PIXEL_GRID_W):\n    \"\"\"\n    Chuyển file .asm thành vector pixel density 1024 chiều.\n    Normalize về [0, 1] bằng cách chia 255.\n    \"\"\"\n    n_cells = grid_h * grid_w\n\n    if not os.path.exists(asm_path):\n        return np.zeros(n_cells, dtype=np.float32)\n\n    try:\n        with open(asm_path, 'rb') as f:\n            raw = np.frombuffer(f.read(), dtype=np.uint8)\n    except Exception:\n        return np.zeros(n_cells, dtype=np.float32)\n\n    if len(raw) == 0:\n        return np.zeros(n_cells, dtype=np.float32)\n\n    # Resample về đúng n_cells điểm bằng linear interpolation\n    # (giữ tỷ lệ vị trí, không padding/crop)\n    indices     = np.linspace(0, len(raw) - 1, n_cells).astype(np.int64)\n    pixel_vec   = raw[indices].astype(np.float32) / 255.0\n\n    return pixel_vec\n\nprint(f\"✅ Pixel density extractor sẵn sàng\")\nprint(f\"   Grid: {PIXEL_GRID_H}×{PIXEL_GRID_W} = {N_PIXEL_FEATURES} features/sample\")","metadata":{"execution":{"iopub.execute_input":"2026-04-13T09:15:26.942546Z","iopub.status.busy":"2026-04-13T09:15:26.942216Z","iopub.status.idle":"2026-04-13T09:15:26.951287Z","shell.execute_reply":"2026-04-13T09:15:26.950218Z"},"papermill":{"duration":0.015849,"end_time":"2026-04-13T09:15:26.953409+00:00","exception":false,"start_time":"2026-04-13T09:15:26.93756+00:00","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ============================================================\n# CELL 8 — HÀM BATCH PROCESSING (TỔNG HỢP TẤT CẢ FEATURES)\n# ============================================================\n\ndef extract_byte_ngram_corpus(batch_ids, temp_dir):\n    \"\"\"\n    Tạo corpus hex strings cho byte N-gram vectorizer.\n    Lọc '??' tokens và chỉ giữ hex bytes hợp lệ.\n    \"\"\"\n    corpus = []\n    for uid in batch_ids:\n        bytes_path = os.path.join(temp_dir, f\"{uid}.bytes\")\n        if os.path.exists(bytes_path):\n            with open(bytes_path, 'r', errors='ignore') as f:\n                # Chỉ lấy token 2 ký tự hex (bỏ '??' và địa chỉ dài)\n                tokens = [w for w in f.read().split() if len(w) == 2 and w != '??']\n                corpus.append(' '.join(tokens))\n        else:\n            corpus.append(\"\")\n    return corpus\n\n\ndef process_batch(batch_ids, archive_path, temp_dir, batch_idx, total_batches):\n    \"\"\"\n    Xử lý 1 batch: giải nén → extract 4 loại features → dọn dẹp.\n    \n    Returns:\n        df_tab    : pd.DataFrame - tabular features\n        ng_sparse : scipy sparse matrix - byte N-gram counts\n        opseq_list: list of str - opcode sequences\n        pixel_mat : np.ndarray (N, 1024) - pixel density features\n    \"\"\"\n    logging.info(f\"[Batch {batch_idx}/{total_batches}] Giải nén {len(batch_ids)} files...\")\n\n    # ── Bước 1: Giải nén batch từ 7z ────────────────────────────────────\n    list_file = os.path.join(WORKING_DIR, 'batch_list.txt')\n    with open(list_file, 'w') as f:\n        for mid in batch_ids:\n            f.write(f\"{mid}.bytes\\n\")\n            f.write(f\"{mid}.asm\\n\")\n\n    cmd = f\"7z e {archive_path} -o{temp_dir} -ir@{list_file} -y -mmt=4\"\n    subprocess.run(cmd, shell=True, stdout=subprocess.DEVNULL, stderr=subprocess.PIPE)\n\n    # ── Bước 2: Tabular features (Multiprocessing) ───────────────────────\n    logging.info(f\"[Batch {batch_idx}/{total_batches}] Tabular features (multiprocessing)...\")\n    args = [(uid, temp_dir) for uid in batch_ids]\n    with Pool(processes=4) as pool:\n        tabular_results = pool.starmap(process_single_sample, args)\n    df_tab = pd.DataFrame(tabular_results)\n\n    # ── Bước 3: Byte N-gram (vectorize) ─────────────────────────────────\n    logging.info(f\"[Batch {batch_idx}/{total_batches}] Byte N-grams (HashingVec 130K)...\")\n    byte_corpus = extract_byte_ngram_corpus(batch_ids, temp_dir)\n    ng_sparse   = BYTE_VECTORIZER.transform(byte_corpus)\n\n    # ── Bước 4: Opcode sequences ─────────────────────────────────────────\n    logging.info(f\"[Batch {batch_idx}/{total_batches}] Opcode sequences...\")\n    opseq_list = []\n    for uid in batch_ids:\n        asm_path = os.path.join(temp_dir, f\"{uid}.asm\")\n        opseq_list.append(extract_opcode_sequence(asm_path))\n\n    # ── Bước 5: ASM pixel density ─────────────────────────────────────────\n    logging.info(f\"[Batch {batch_idx}/{total_batches}] ASM pixel density (1024 features)...\")\n    pixel_mat = np.vstack([\n        extract_pixel_density(os.path.join(temp_dir, f\"{uid}.asm\"))\n        for uid in batch_ids\n    ])\n\n    # ── Bước 6: Dọn dẹp file tạm ─────────────────────────────────────────\n    logging.info(f\"[Batch {batch_idx}/{total_batches}] Dọn dẹp, giải phóng RAM...\")\n    for mid in batch_ids:\n        for ext in ('.bytes', '.asm'):\n            p = os.path.join(temp_dir, f\"{mid}{ext}\")\n            if os.path.exists(p):\n                os.remove(p)\n    gc.collect()\n\n    logging.info(f\"[Batch {batch_idx}/{total_batches}] ✓ Hoàn tất batch\")\n    logging.info(\"-\" * 55)\n\n    return df_tab, ng_sparse, opseq_list, pixel_mat\n\n\ndef run_pipeline(ids_list, archive_path, phase_name):\n    \"\"\"\n    Chạy pipeline cho toàn bộ danh sách IDs,\n    trả về 4 cấu trúc dữ liệu đã ghép nối.\n    \"\"\"\n    total_batches = math.ceil(len(ids_list) / BATCH_SIZE)\n    logging.info(\n        f\"=== BẮT ĐẦU {phase_name.upper()} \"\n        f\"({len(ids_list)} mẫu | {total_batches} batches) ===\"\n    )\n\n    all_tab, all_ng, all_opseq, all_pixel = [], [], [], []\n\n    for i in range(0, len(ids_list), BATCH_SIZE):\n        batch_ids = ids_list[i : i + BATCH_SIZE]\n        idx       = i // BATCH_SIZE + 1\n\n        df_tab, ng_sparse, opseq_list, pixel_mat = \\\n            process_batch(batch_ids, archive_path, TEMP_DIR, idx, total_batches)\n\n        all_tab.append(df_tab)\n        all_ng.append(ng_sparse)\n        all_opseq.extend(opseq_list)     # list extend (không stack)\n        all_pixel.append(pixel_mat)\n\n    logging.info(f\"=== GHÉP NỐI {phase_name.upper()} ===\")\n    final_tab   = pd.concat(all_tab, ignore_index=True)\n    final_ng    = sp.vstack(all_ng)\n    final_pixel = np.vstack(all_pixel)\n\n    return final_tab, final_ng, all_opseq, final_pixel\n\nprint(\"✅ Batch processing functions sẵn sàng\")","metadata":{"execution":{"iopub.execute_input":"2026-04-13T09:15:26.962904Z","iopub.status.busy":"2026-04-13T09:15:26.962553Z","iopub.status.idle":"2026-04-13T09:15:26.980267Z","shell.execute_reply":"2026-04-13T09:15:26.979145Z"},"papermill":{"duration":0.02528,"end_time":"2026-04-13T09:15:26.982355+00:00","exception":false,"start_time":"2026-04-13T09:15:26.957075+00:00","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ============================================================\n# CELL 9 — LOAD DANH SÁCH ID & CHIA PHẦN\n# ============================================================\n\n# Đọc danh sách IDs dựa trên TARGET\nif TARGET == \"TRAIN\":\n    df_labels    = pd.read_csv(LABEL_CSV)\n    full_ids     = df_labels['Id'].tolist()\n    y_full       = df_labels['Class'].values - 1  # về 0-8\n    archive_path = TRAIN_7Z\nelif TARGET == \"TEST\":\n    df_sub       = pd.read_csv(SUBMISSION_CSV)\n    full_ids     = df_sub['Id'].tolist()\n    archive_path = TEST_7Z\nelse:\n    raise ValueError(\"TARGET phải là 'TRAIN' hoặc 'TEST'\")\n\n# Chia đều danh sách thành TOTAL_PARTS phần\nchunk_size = math.ceil(len(full_ids) / TOTAL_PARTS)\nstart_idx  = (PART - 1) * chunk_size\nend_idx    = min(start_idx + chunk_size, len(full_ids))\ntarget_ids = full_ids[start_idx : end_idx]\n\nif TARGET == \"TRAIN\":\n    y_target = y_full[start_idx : end_idx]\n\nlogging.info(\n    f\"🚀 {TARGET} PART {PART}/{TOTAL_PARTS}: \"\n    f\"index [{start_idx}..{end_idx-1}] — {len(target_ids)} mẫu\"\n)\nprint(f\"📦 Số mẫu cần xử lý: {len(target_ids)}\")","metadata":{"execution":{"iopub.execute_input":"2026-04-13T09:15:26.991603Z","iopub.status.busy":"2026-04-13T09:15:26.991236Z","iopub.status.idle":"2026-04-13T09:15:27.064505Z","shell.execute_reply":"2026-04-13T09:15:27.063495Z"},"papermill":{"duration":0.080323,"end_time":"2026-04-13T09:15:27.066424+00:00","exception":false,"start_time":"2026-04-13T09:15:26.986101+00:00","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ============================================================\n# CELL 10 — CHẠY PIPELINE CHÍNH\n# ============================================================\n\ndf_data, X_ng_sparse, opseq_corpus, X_pixel = run_pipeline(\n    target_ids, archive_path, f\"{TARGET} Part {PART}\"\n)\n\n# Tách tabular thành numpy array (bỏ cột Id)\nX_tab = df_data.drop(columns=['Id']).fillna(0).values.astype(np.float32)\n\nprint(f\"\\n📊 Kích thước kết quả:\")\nprint(f\"  Tabular  : {X_tab.shape}\")\nprint(f\"  Byte N-g : {X_ng_sparse.shape}\")\nprint(f\"  Opcode seq: {len(opseq_corpus)} strings\")\nprint(f\"  Pixel    : {X_pixel.shape}\")","metadata":{"execution":{"iopub.execute_input":"2026-04-13T09:15:27.07571Z","iopub.status.busy":"2026-04-13T09:15:27.075361Z","iopub.status.idle":"2026-04-13T17:47:48.972483Z","shell.execute_reply":"2026-04-13T17:47:48.971073Z"},"papermill":{"duration":30741.906238,"end_time":"2026-04-13T17:47:48.976539+00:00","exception":false,"start_time":"2026-04-13T09:15:27.070301+00:00","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ============================================================\n# CELL 11 — LƯU KẾT QUẢ\n# ============================================================\n\nprefix = f\"{TARGET.lower()}_part{PART}\"\n\n# 1. Tabular features\ntab_path = os.path.join(WORKING_DIR, f'X_{prefix}_tab.npy')\nnp.save(tab_path, X_tab)\nlogging.info(f\"💾 Saved tabular: {tab_path}  shape={X_tab.shape}\")\n\n# 2. Byte N-gram sparse matrix\nng_path = os.path.join(WORKING_DIR, f'X_{prefix}_ng.npz')\nsp.save_npz(ng_path, X_ng_sparse)\nlogging.info(f\"💾 Saved byte N-gram: {ng_path}  shape={X_ng_sparse.shape}\")\n\n# 3. Opcode sequences (list of strings)\nopseq_path = os.path.join(WORKING_DIR, f'X_{prefix}_opseq.pkl')\nwith open(opseq_path, 'wb') as f:\n    pickle.dump(opseq_corpus, f)\nlogging.info(f\"💾 Saved opcode seqs: {opseq_path}  n={len(opseq_corpus)}\")\n\n# 4. Pixel density features\npixel_path = os.path.join(WORKING_DIR, f'X_{prefix}_pixel.npy')\nnp.save(pixel_path, X_pixel)\nlogging.info(f\"💾 Saved pixel: {pixel_path}  shape={X_pixel.shape}\")\n\n# 5. Labels (chỉ khi TRAIN)\nif TARGET == \"TRAIN\":\n    y_path = os.path.join(WORKING_DIR, f'y_{prefix}.npy')\n    np.save(y_path, y_target)\n    logging.info(f\"💾 Saved labels: {y_path}  shape={y_target.shape}\")\n\nprint(f\"\\n✅ HOÀN TẤT PART {PART}/{TOTAL_PARTS}!\")\nprint(f\"📁 Tất cả files tại: {WORKING_DIR}\")\nprint(\"\\n📋 Files cần upload lên Kaggle Dataset:\")\nfiles = [\n    f'X_{prefix}_tab.npy',\n    f'X_{prefix}_ng.npz',\n    f'X_{prefix}_opseq.pkl',\n    f'X_{prefix}_pixel.npy',\n]\nif TARGET == \"TRAIN\":\n    files.append(f'y_{prefix}.npy')\nfor fn in files:\n    fp = os.path.join(WORKING_DIR, fn)\n    size_mb = os.path.getsize(fp) / 1e6 if os.path.exists(fp) else 0\n    print(f\"  {fn:<45} {size_mb:>6.1f} MB\")","metadata":{"execution":{"iopub.execute_input":"2026-04-13T17:47:49.020354Z","iopub.status.busy":"2026-04-13T17:47:49.019784Z","iopub.status.idle":"2026-04-13T17:50:51.448418Z","shell.execute_reply":"2026-04-13T17:50:51.447054Z"},"papermill":{"duration":182.458513,"end_time":"2026-04-13T17:50:51.451844+00:00","exception":false,"start_time":"2026-04-13T17:47:48.993331+00:00","status":"completed"},"tags":[]},"outputs":[],"execution_count":null}]}