{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceType":"competition","sourceId":4117,"databundleVersionId":46665},{"sourceType":"datasetVersion","sourceId":15870212,"datasetId":10174685,"databundleVersionId":16822892}],"dockerImageVersionId":31329,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport time\nimport shutil\nimport subprocess\nimport pandas as pd\nimport multiprocessing as mp\nfrom functools import partial\n\nprint(\"=\"*60)\nprint(\"KHỞI TẠO TIẾN TRÌNH TRÍCH XUẤT\")\nprint(\"=\"*60)\n\n# 1. TÌM FILE TEST.7Z \nTEST_ARCHIVE = None\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    if 'test.7z' in filenames:\n        TEST_ARCHIVE = os.path.join(dirname, 'test.7z')\n        break\n\nif not TEST_ARCHIVE:\n    raise FileNotFoundError(\"[!] KHÔNG TÌM THẤY test.7z.\")\n\nTEMP_DIR = '/kaggle/working/temp_test_extract'\nif os.path.exists(TEMP_DIR):\n    shutil.rmtree(TEMP_DIR)\nos.makedirs(TEMP_DIR, exist_ok=True)\n\nprint(\"[*] Đang lập bản đồ cấu trúc file nén...\")\nresult = subprocess.run(['7z', 'l', TEST_ARCHIVE], capture_output=True, text=True)\nlines = result.stdout.split('\\n')\n\n# Lấy chính xác đường dẫn của từng file bên trong cục nén\narchive_bytes = [line.strip().split()[-1] for line in lines if line.strip().endswith('.bytes')]\narchive_asm = [line.strip().split()[-1] for line in lines if line.strip().endswith('.asm')]\n\nid_to_paths = {}\nfor p in archive_bytes:\n    fid = p.split('/')[-1].replace('.bytes', '')\n    if fid not in id_to_paths: id_to_paths[fid] = {}\n    id_to_paths[fid]['bytes'] = p\n    \nfor p in archive_asm:\n    fid = p.split('/')[-1].replace('.asm', '')\n    if fid in id_to_paths: id_to_paths[fid]['asm'] = p\n\ntest_ids = list(id_to_paths.keys())\nprint(f\" -> Đã chốt sổ danh sách: {len(test_ids)} file.\")\n\n# 3. ĐỊNH NGHĨA KHÔNG GIAN ĐẶC TRƯNG \nbyte_cols = [f'byte_{hex(i)[2:].upper().zfill(2)}' for i in range(256)] + ['byte_??']\nopcodes = ['add', 'al', 'bt', 'call', 'cdq', 'cld', 'cli', 'cmc', 'cmp', 'cwd', 'daa', 'das', 'dec', 'div', 'hlt', 'idiv', 'imul', 'inc', 'int', 'int3', 'into', 'iret', 'ja', 'jae', 'jb', 'jbe', 'jc', 'jcxz', 'je', 'jecxz', 'jg', 'jge', 'jl', 'jle', 'jmp', 'jna', 'jnae', 'jnb', 'jnbe', 'jnc', 'jne', 'jng', 'jnge', 'jnl', 'jnle', 'jno', 'jnp', 'jns', 'jnz', 'jo', 'jp', 'jpe', 'jpo', 'js', 'jz', 'lea', 'lock', 'lods', 'loop', 'loope', 'loopne', 'loopnz', 'loopz', 'mov', 'movs', 'movsx', 'movzx', 'mul', 'neg', 'nop', 'not', 'or', 'out', 'outs', 'pop', 'popa', 'popad', 'popf', 'popfd', 'push', 'pusha', 'pushad', 'pushf', 'pushfd', 'rcl', 'rcr', 'rep', 'repe', 'repne', 'repnz', 'repz', 'ret', 'retf', 'retn', 'rol', 'ror', 'sahf', 'sal', 'sar', 'sbb', 'scas', 'seta', 'setae', 'setb', 'setbe', 'setc', 'sete', 'setg', 'setge', 'setl', 'setle', 'setna', 'setnae', 'setnb', 'setnbe', 'setnc', 'setne', 'setng', 'setnge', 'setnl', 'setnle', 'setno', 'setnp', 'setns', 'setnz', 'seto', 'setp', 'setpe', 'setpo', 'sets', 'setz', 'shl', 'shld', 'shr', 'shrd', 'stc', 'std', 'sti', 'stos', 'sub', 'test', 'wait', 'xchg', 'xlat', 'xor']\nsections = ['.text', '.data', '.rdata', '.bss', '.idata', '.edata', '.rsrc', '.tls', '.reloc']\ncolumns = ['ID', 'Size_Bytes', 'Size_ASM'] + byte_cols + [f'op_{op}' for op in opcodes] + [f'sec_{sec[1:]}' for sec in sections] \n\nopcode_set = set(opcodes)\nsection_set = set(sections)\n\n# 4. HÀM XỬ LÝ LÕI\ndef extract_test_features(file_id, cols, temp_dir):\n    features = {col: 0 for col in cols}\n    features['ID'] = file_id\n    \n    bytes_path = os.path.join(temp_dir, f\"{file_id}.bytes\")\n    asm_path = os.path.join(temp_dir, f\"{file_id}.asm\")\n    \n    if os.path.exists(bytes_path):\n        features['Size_Bytes'] = os.path.getsize(bytes_path)\n        with open(bytes_path, 'r', encoding='utf-8', errors='ignore') as f:\n            for line in f:\n                tokens = line.strip().split()[1:]\n                for t in tokens:\n                    if t == '??': features['byte_??'] += 1\n                    else:\n                        col = f'byte_{t}'\n                        if col in features: features[col] += 1\n                        \n    if os.path.exists(asm_path):\n        features['Size_ASM'] = os.path.getsize(asm_path)\n        with open(asm_path, 'r', encoding='utf-8', errors='ignore') as f:\n            for line in f:\n                line_lower = line.lower().strip()\n                words = line_lower.split()\n                for word in words:\n                    if word in opcode_set: features[f'op_{word}'] += 1\n                for sec in sections:\n                    if line_lower.startswith(sec): features[f'sec_{sec[1:]}'] += 1\n    return features\n\n# 5. KÍCH HOẠT ĐA LUỒNG\nif __name__ == '__main__':\n    num_cores = mp.cpu_count()\n    BATCH_SIZE = 200 \n    total_files = len(test_ids)\n    all_results = []\n    \n    print(f\"[*] Bắt đầu xử lý {total_files} file với {num_cores} luồng CPU...\")\n    start_time = time.time()\n    \n    extract_target = os.path.join(TEMP_DIR, 'test_batch')\n    worker_func = partial(extract_test_features, cols=columns, temp_dir=extract_target)\n    \n    for i in range(0, total_files, BATCH_SIZE):\n        batch_ids = test_ids[i : i + BATCH_SIZE]\n        \n        # Ghi file theo đúng đường dẫn gốc đã map ở trên\n        listfile_path = os.path.join('/kaggle/working', 'batch_list.txt')\n        with open(listfile_path, 'w') as f:\n            for fid in batch_ids:\n                if 'bytes' in id_to_paths[fid]: f.write(f\"{id_to_paths[fid]['bytes']}\\n\")\n                if 'asm' in id_to_paths[fid]: f.write(f\"{id_to_paths[fid]['asm']}\\n\")\n        \n        # Đảm bảo thư mục đích tồn tại\n        if os.path.exists(extract_target):\n            shutil.rmtree(extract_target)\n        os.makedirs(extract_target, exist_ok=True)\n        \n        # Gọi lệnh e để giải nén (bung phẳng cấu trúc thư mục)\n        cmd = ['7z', 'e', TEST_ARCHIVE, f'-o{extract_target}', f'@{listfile_path}', '-y']\n        subprocess.run(cmd, stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL)\n        \n        # --- BÁO ĐỘNG ĐỎ (SANITY CHECK) ---\n        extracted_files = os.listdir(extract_target)\n        if len(extracted_files) == 0:\n            raise RuntimeError(\"[!] BÁO ĐỘNG ĐỎ: 7z giải nén thất bại! Không có file nào nằm trong đĩa.\")\n        \n        # Chạy đa luồng tính toán\n        with mp.Pool(num_cores) as pool:\n            batch_data = pool.map(worker_func, batch_ids)\n            all_results.extend(batch_data)\n        \n        elapsed = (time.time() - start_time) / 60\n        print(f\" -> Đã quét: {min(i + BATCH_SIZE, total_files)}/{total_files} | Thời gian: {elapsed:.2f} phút\")\n\n    # Dọn rác\n    shutil.rmtree(TEMP_DIR)\n\n    final_test_df = pd.DataFrame(all_results, columns=columns)\n    output_filename = 'REAL_TEST_FEATURES_KAGGLE.csv'\n    final_test_df.to_csv(output_filename, index=False)\n    print(f\"\\nĐã lưu trữ thành công file: {output_filename}\")","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}