{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceType":"competition","sourceId":4117,"databundleVersionId":46665}],"dockerImageVersionId":31328,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport time\nimport shutil\nimport subprocess\nimport pandas as pd\nimport multiprocessing as mp\nfrom functools import partial\nfrom sklearn.model_selection import train_test_split\n\nprint(\"--- BƯỚC 1: TIỀN XỬ LÝ NHÃN VÀ CHIA TẬP 70-15-15 ---\")\n\n# 1. Đường dẫn gốc từ Kaggle\nINPUT_DIR = '/kaggle/input/competitions/malware-classification'\nTRAIN_ARCHIVE = os.path.join(INPUT_DIR, 'train.7z')\nLABEL_PATH = os.path.join(INPUT_DIR, 'trainLabels.csv')\n\n# Thư mục tạm để giải nén (Sẽ liên tục bị xóa và tạo lại)\nTEMP_DIR = '/kaggle/working/temp_extract'\nos.makedirs(TEMP_DIR, exist_ok=True)\n\n# 2. Đọc và lọc nhãn (Chuẩn mực Data Leakage của nhóm)\ndf_labels = pd.read_csv(LABEL_PATH)\ndf_labels = df_labels.rename(columns={'Id': 'ID'})\ndf_labels = df_labels.drop_duplicates(subset=['ID']).dropna(subset=['Class'])\ndf_labels = df_labels[df_labels['Class'].isin(range(1, 10))]\n\n# 3. Chia tập\ntrain_df, temp_df = train_test_split(df_labels, test_size=0.3, stratify=df_labels['Class'], random_state=42)\nval_df, test_df = train_test_split(temp_df, test_size=0.5, stratify=temp_df['Class'], random_state=42)\n\n# Hash sets để điều hướng siêu tốc\ntrain_ids, val_ids, test_ids = set(train_df['ID']), set(val_df['ID']), set(test_df['ID'])\nprint(f\"✅ Đã chia tập: Train ({len(train_ids)}), Val ({len(val_ids)}), Test ({len(test_ids)})\")\n\n# 4. Danh sách 400+ Features\nbyte_cols = [f'byte_{hex(i)[2:].upper().zfill(2)}' for i in range(256)] + ['byte_??']\nopcodes = ['add', 'al', 'bt', 'call', 'cdq', 'cld', 'cli', 'cmc', 'cmp', 'cwd', 'daa', 'das', 'dec', 'div', 'hlt', 'idiv', 'imul', 'inc', 'int', 'int3', 'into', 'iret', 'ja', 'jae', 'jb', 'jbe', 'jc', 'jcxz', 'je', 'jecxz', 'jg', 'jge', 'jl', 'jle', 'jmp', 'jna', 'jnae', 'jnb', 'jnbe', 'jnc', 'jne', 'jng', 'jnge', 'jnl', 'jnle', 'jno', 'jnp', 'jns', 'jnz', 'jo', 'jp', 'jpe', 'jpo', 'js', 'jz', 'lea', 'lock', 'lods', 'loop', 'loope', 'loopne', 'loopnz', 'loopz', 'mov', 'movs', 'movsx', 'movzx', 'mul', 'neg', 'nop', 'not', 'or', 'out', 'outs', 'pop', 'popa', 'popad', 'popf', 'popfd', 'push', 'pusha', 'pushad', 'pushf', 'pushfd', 'rcl', 'rcr', 'rep', 'repe', 'repne', 'repnz', 'repz', 'ret', 'retf', 'retn', 'rol', 'ror', 'sahf', 'sal', 'sar', 'sbb', 'scas', 'seta', 'setae', 'setb', 'setbe', 'setc', 'sete', 'setg', 'setge', 'setl', 'setle', 'setna', 'setnae', 'setnb', 'setnbe', 'setnc', 'setne', 'setng', 'setnge', 'setnl', 'setnle', 'setno', 'setnp', 'setns', 'setnz', 'seto', 'setp', 'setpe', 'setpo', 'sets', 'setz', 'shl', 'shld', 'shr', 'shrd', 'stc', 'std', 'sti', 'stos', 'sub', 'test', 'wait', 'xchg', 'xlat', 'xor']\nsections = ['.text', '.data', '.rdata', '.bss', '.idata', '.edata', '.rsrc', '.tls', '.reloc']\ncolumns = ['ID', 'Size_Bytes', 'Size_ASM'] + byte_cols + [f'op_{op}' for op in opcodes] + [f'sec_{sec[1:]}' for sec in sections] + ['Class', 'Split_Group']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-10T16:27:23.65255Z","iopub.execute_input":"2026-04-10T16:27:23.652996Z","iopub.status.idle":"2026-04-10T16:27:23.667764Z","shell.execute_reply.started":"2026-04-10T16:27:23.652962Z","shell.execute_reply":"2026-04-10T16:27:23.666272Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"--- BƯỚC 2: HÀM TRÍCH XUẤT TỐI ƯU TỐC ĐỘ (TOKENIZATION) ---\")\n\n# Chuyển opcodes và sections thành cấu trúc Set để tra cứu với tốc độ O(1)\nopcode_set = set(opcodes)\nsection_set = set(sections)\n\ndef process_extracted_file(file_info, cols, temp_dir, train_set, val_set):\n    file_id, label = file_info\n    features = {col: 0 for col in cols}\n    features['ID'] = file_id\n    features['Class'] = label\n    \n    if file_id in train_set: features['Split_Group'] = 'Train'\n    elif file_id in val_set: features['Split_Group'] = 'Val'\n    else: features['Split_Group'] = 'Test'\n    \n    bytes_path = os.path.join(temp_dir, f\"{file_id}.bytes\")\n    asm_path = os.path.join(temp_dir, f\"{file_id}.asm\")\n    \n    # 1. Xử lý file .bytes\n    if os.path.exists(bytes_path):\n        features['Size_Bytes'] = os.path.getsize(bytes_path)\n        with open(bytes_path, 'r', encoding='utf-8', errors='ignore') as f:\n            for line in f:\n                tokens = line.strip().split()[1:]\n                for t in tokens:\n                    if t == '??': features['byte_??'] += 1\n                    else:\n                        col_name = f'byte_{t}'\n                        if col_name in features: features[col_name] += 1\n                        \n    # 2. Xử lý file .asm (THUẬT TOÁN ĐÃ ĐƯỢC ĐỘ LẠI)\n    if os.path.exists(asm_path):\n        features['Size_ASM'] = os.path.getsize(asm_path)\n        with open(asm_path, 'r', encoding='utf-8', errors='ignore') as f:\n            for line in f:\n                line_lower = line.lower().strip()\n                \n                # Cắt dòng thành các từ và kiểm tra ngay lập tức trong Set\n                words = line_lower.split()\n                for word in words:\n                    if word in opcode_set:\n                        features[f'op_{word}'] += 1\n                \n                # Kiểm tra Section ở đầu dòng\n                for sec in sections:\n                    if line_lower.startswith(sec):\n                        features[f'sec_{sec[1:]}'] += 1\n                    \n    return features","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import shutil\nimport subprocess\nimport time\nimport os\nimport multiprocessing as mp\nfrom functools import partial\nimport pandas as pd\n\nprint(\"--- BƯỚC 3: GIẢI NÉN CHUNKS VÀ XỬ LÝ ĐA LUỒNG ---\")\n\nif __name__ == '__main__':\n    file_info_list = list(zip(df_labels['ID'], df_labels['Class']))\n    num_cores = mp.cpu_count()\n    BATCH_SIZE = 200 \n    total_files = len(file_info_list)\n    batch_count = 1\n    start_time = time.time()\n    \n    worker_func = partial(\n        process_extracted_file, cols=columns, temp_dir=TEMP_DIR, \n        train_set=train_ids, val_set=val_ids\n    )\n\n    for i in range(0, total_files, BATCH_SIZE):\n        batch_info = file_info_list[i : i + BATCH_SIZE]\n        \n        listfile_path = os.path.join('/kaggle/working', 'listfile.txt')\n        with open(listfile_path, 'w') as f:\n            for file_id, _ in batch_info:\n                f.write(f\"train/{file_id}.bytes\\n\")\n                f.write(f\"train/{file_id}.asm\\n\")\n                \n        cmd = ['7z', 'e', TRAIN_ARCHIVE, f'-o{TEMP_DIR}', f'@{listfile_path}', '-y']\n        subprocess.run(cmd, stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL)\n        \n        with mp.Pool(num_cores) as pool:\n            batch_results = pool.map(worker_func, batch_info)\n            \n        df_batch = pd.DataFrame(batch_results, columns=columns)\n        \n        # Sử dụng tên mới ở đây\n        batch_filename = f'temp_batch_data_{batch_count}.csv'\n        df_batch.to_csv(batch_filename, index=False)\n        \n        shutil.rmtree(TEMP_DIR)\n        os.makedirs(TEMP_DIR, exist_ok=True)\n        \n        elapsed = (time.time() - start_time) / 60\n        print(f\"💾 Xong lô {batch_count} | Tiến độ: {min(i + BATCH_SIZE, total_files)}/{total_files} | Thời gian: {elapsed:.2f} phút\")\n        batch_count += 1","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import glob\nimport pandas as pd\n\nprint(\"--- BƯỚC 4: TỔNG HỢP VÀ PHÂN CHIA TRAIN/VAL/TEST ---\")\n\n# Quét đúng tên file mới\nall_batch_files = glob.glob('temp_batch_data_*.csv')\ndf_list = [pd.read_csv(f) for f in all_batch_files]\ndf_final = pd.concat(df_list, axis=0).reset_index(drop=True)\n\ndf_train = df_final[df_final['Split_Group'] == 'Train'].drop(columns=['Split_Group'])\ndf_val = df_final[df_final['Split_Group'] == 'Val'].drop(columns=['Split_Group'])\ndf_test = df_final[df_final['Split_Group'] == 'Test'].drop(columns=['Split_Group'])\n\n# Lưu 3 file chuẩn xác\ndf_train.to_csv('train.csv', index=False)\ndf_val.to_csv('val.csv', index=False)\ndf_test.to_csv('test.csv', index=False)\n\nprint(\"🎉 XUẤT XƯỞNG THÀNH CÔNG!\")\nprint(f\" Train: {df_train.shape} | Val: {df_val.shape} | Test: {df_test.shape}\")","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}