{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceType":"competition","sourceId":4117,"databundleVersionId":46665},{"sourceType":"datasetVersion","sourceId":15878877,"datasetId":10180842,"databundleVersionId":16832241}],"dockerImageVersionId":31328,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nsample_sub = pd.read_csv('/kaggle/input/competitions/malware-classification/sampleSubmission.csv')\ntest_ids = sample_sub['Id'].tolist()\n# Tổng cộng khoảng 10.873 file, mỗi máy sẽ thầu ~2.720 file\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-23T08:03:21.265200Z","iopub.execute_input":"2026-04-23T08:03:21.265851Z","iopub.status.idle":"2026-04-23T08:03:21.330947Z","shell.execute_reply.started":"2026-04-23T08:03:21.265819Z","shell.execute_reply":"2026-04-23T08:03:21.330007Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport pandas as pd\nimport numpy as np\nimport subprocess\nimport gc\n\n# ==========================================\n# 1. CẤU HÌNH ĐOẠN CHẠY (SỬA Ở ĐÂY CHO MỖI MÁY)\n# ==========================================\n# Máy 1: START, END = 0, 2720    | PART_NAME = 'test_part1.csv'\n# Máy 2: START, END = 2720, 5440 | PART_NAME = 'test_part2.csv'\n# Máy 3: START, END = 5440, 8160 | PART_NAME = 'test_part3.csv'\n# Máy 4: START, END = 8160, None | PART_NAME = 'test_part4.csv'\n\nSTART, END = 8160, None  # <--- SỬA DÒNG NÀY THEO TỪNG MÁY\nPART_NAME = 'test_part4.csv' # <--- SỬA TÊN FILE TƯƠNG ỨNG\n\n# ==========================================\n# 2. BIẾN CẤU HÌNH HỆ THỐNG\n# ==========================================\nARCHIVE_PATH = '/kaggle/input/competitions/malware-classification/test.7z'\nTEMP_DIR = f'/kaggle/working/temp_{PART_NAME.split(\".\")[0]}'\nBATCH_SIZE = 150 \nTARGET_OPCODES = ['mov', 'push', 'pop', 'jmp', 'call', 'ret', 'cmp', 'add', 'sub', 'xor', 'inc', 'dec', 'lea']\ncolumns = [format(j, '02x').upper() for j in range(256)] + ['??'] + TARGET_OPCODES + ['size_bytes', 'size_asm']\n\nos.makedirs(TEMP_DIR, exist_ok=True)\n\n# ==========================================\n# 3. ĐỊNH NGHĨA HÀM XỬ LÝ (GET FEATURES)\n# ==========================================\ndef get_features_from_disk(f_id):\n    # Trích xuất .bytes\n    byte_counts = np.zeros(257, dtype=int)\n    byte_path = os.path.join(TEMP_DIR, f\"{f_id}.bytes\")\n    size_bytes = os.path.getsize(byte_path) if os.path.exists(byte_path) else 0\n    if os.path.exists(byte_path):\n        with open(byte_path, 'r') as f:\n            for line in f:\n                tokens = line.strip().split()\n                for hx in tokens[1:]:\n                    if hx == '??': byte_counts[256] += 1\n                    else:\n                        try: byte_counts[int(hx, 16)] += 1\n                        except: pass\n    \n    # Trích xuất .asm\n    opcode_counts = {op: 0 for op in TARGET_OPCODES}\n    asm_path = os.path.join(TEMP_DIR, f\"{f_id}.asm\")\n    size_asm = os.path.getsize(asm_path) if os.path.exists(asm_path) else 0\n    if os.path.exists(asm_path):\n        with open(asm_path, 'r', encoding='latin-1') as f:\n            for line in f:\n                tokens = line.strip().split()\n                for t in tokens:\n                    cl = t.lower()\n                    if cl in opcode_counts: opcode_counts[cl] += 1\n                    \n    return list(byte_counts) + list(opcode_counts.values()) + [size_bytes, size_asm]\n\n# ==========================================\n# 4. CHƯƠNG TRÌNH CHÍNH\n# ==========================================\nprint(f\"🚀 Đang nạp danh sách file cho {PART_NAME}...\")\nsample_sub = pd.read_csv('/kaggle/input/competitions/malware-classification/sampleSubmission.csv')\ntest_ids_all = sample_sub['Id'].tolist()\n\n# Lấy đoạn ID tương ứng cho máy này\nif END is None:\n    current_batch_ids = test_ids_all[START:]\nelse:\n    current_batch_ids = test_ids_all[START:END]\n\nall_data = []\n\nprint(f\"📦 Bắt đầu xử lý {len(current_batch_ids)} file từ vị trí {START}...\")\n\nfor i in range(0, len(current_batch_ids), BATCH_SIZE):\n    batch = current_batch_ids[i : i + BATCH_SIZE]\n    \n    # Tạo danh sách file để giải nén cụm\n    files_to_extract = []\n    for f_id in batch:\n        files_to_extract.extend([f\"{f_id}.bytes\", f\"{f_id}.asm\"])\n    \n    # Lệnh giải nén 7z (Batch)\n    cmd = ['7z', 'e', ARCHIVE_PATH] + files_to_extract + ['-o' + TEMP_DIR, '-y', '-r']\n    subprocess.run(cmd, stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL)\n    \n    # Xử lý từng file sau khi bung\n    for f_id in batch:\n        features = get_features_from_disk(f_id)\n        all_data.append(features)\n        \n        # Xóa ngay để dọn ổ cứng\n        try:\n            os.remove(os.path.join(TEMP_DIR, f\"{f_id}.bytes\"))\n            os.remove(os.path.join(TEMP_DIR, f\"{f_id}.asm\"))\n        except: pass\n    \n    gc.collect()\n    if (i + len(batch)) % 300 == 0 or (i + len(batch)) == len(current_batch_ids):\n        print(f\"✅ Đã hoàn thành: {i + len(batch)}/{len(current_batch_ids)} file\")\n\n# ==========================================\n# 5. LƯU KẾT QUẢ (KHÔNG CÒN LỖI NAMEERROR)\n# ==========================================\nprint(f\"💾 Đang đóng gói dữ liệu vào {PART_NAME}...\")\ndf_part = pd.DataFrame(all_data, columns=columns)\ndf_part.insert(0, 'Id', current_batch_ids)\ndf_part.to_csv(PART_NAME, index=False)\n\nprint(f\"✨ HOÀN TẤT! File {PART_NAME} đã sẵn sàng.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-23T08:03:34.199935Z","iopub.execute_input":"2026-04-23T08:03:34.200642Z","iopub.status.idle":"2026-04-23T08:03:46.078842Z","shell.execute_reply.started":"2026-04-23T08:03:34.200602Z","shell.execute_reply":"2026-04-23T08:03:46.077513Z"}},"outputs":[],"execution_count":null}]}