{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"codemirror_mode":{"name":"ipython","version":3},"file_extension":".py","mimetype":"text/x-python","name":"python","nbconvert_exporter":"python","pygments_lexer":"ipython3","version":"3.12.12"},"kaggle":{"accelerator":"none","dataSources":[{"databundleVersionId":46665,"sourceId":4117,"sourceType":"competition"}],"dockerImageVersionId":31328,"isGpuEnabled":false,"isInternetEnabled":true,"language":"python","sourceType":"notebook"},"papermill":{"default_parameters":{},"duration":29200.85362,"end_time":"2026-04-08T09:02:56.532038+00:00","environment_variables":{},"exception":null,"input_path":"__notebook__.ipynb","output_path":"__notebook__.ipynb","parameters":{},"start_time":"2026-04-08T00:56:15.678418+00:00","version":"2.7.0"}},"nbformat_minor":4,"nbformat":4,"cells":[{"id":"43da9216","cell_type":"code","source":"import os\nimport re\nimport gc\nimport zlib\nimport subprocess\nimport numpy as np\nimport pandas as pd\nimport scipy.sparse as sp\nfrom tqdm.notebook import tqdm\nfrom multiprocessing import Pool\nfrom sklearn.feature_extraction.text import HashingVectorizer\nfrom sklearn.feature_selection import SelectKBest, f_classif\nfrom scipy.stats import entropy\n\n# ==========================================\n# CẤU HÌNH THÔNG SỐ CHUẨN (TỪ NOTEBOOK)\n# ==========================================\nDATA_DIR = '/kaggle/input/competitions/malware-classification'\nTRAIN_7Z = os.path.join(DATA_DIR, 'train.7z')\nTEST_7Z = os.path.join(DATA_DIR, 'test.7z')\nLABEL_CSV = os.path.join(DATA_DIR, 'trainLabels.csv')\nSUBMISSION_CSV = os.path.join(DATA_DIR, 'sampleSubmission.csv')\n\nWORKING_DIR = '/kaggle/working'\nTEMP_DIR = os.path.join(WORKING_DIR, 'temp_unzip')\nOUT_DIR = os.path.join(WORKING_DIR, 'processed_data')\n\nos.makedirs(TEMP_DIR, exist_ok=True)\nos.makedirs(OUT_DIR, exist_ok=True)\n\nBATCH_SIZE = 50 # Số lượng file giải nén mỗi lượt\n\n# ==========================================\n# KHỞI TẠO DỮ LIỆU\n# ==========================================\ntrain_labels_df = pd.read_csv(LABEL_CSV)\ntrain_ids = train_labels_df['Id'].tolist()\ny_train = train_labels_df['Class'].values - 1  # Chuyển nhãn 1-9 về 0-8\n\nsub_df = pd.read_csv(SUBMISSION_CSV)\ntest_ids = sub_df['Id'].tolist()\n\nnp.save(os.path.join(OUT_DIR, 'y_train.npy'), y_train)","metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","execution":{"iopub.execute_input":"2026-04-08T00:56:19.554177Z","iopub.status.busy":"2026-04-08T00:56:19.553844Z","iopub.status.idle":"2026-04-08T00:56:23.699785Z","shell.execute_reply":"2026-04-08T00:56:23.698835Z"},"papermill":{"duration":4.152253,"end_time":"2026-04-08T00:56:23.70193+00:00","exception":false,"start_time":"2026-04-08T00:56:19.549677+00:00","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"2f0e5c46","cell_type":"code","source":"OPCODES = ['jmp', 'mov', 'push', 'pop', 'xor', 'call', 'add', 'sub', 'inc', 'dec', 'cmp', 'test']\nSECTIONS = ['HEADER', '.text', '.data', '.rdata', '.bss', '.idata', '.edata']\nINTERPUNCTIONS = [',', '-', '\\\\[', '\\\\]', '\\\\*']\n\ndef calculate_entropy(data):\n    # Sửa lỗi: dùng len() để kiểm tra độ dài thay vì kiểm tra truth value trực tiếp\n    if len(data) == 0: \n        return 0\n    # Chuyển data thành mảng để tính histogram nhanh hơn\n    p, _ = np.histogram(data, bins=256, range=(0, 255), density=True)\n    return entropy(p + 1e-10)\n\ndef extract_file_properties(file_path):\n    try:\n        with open(file_path, 'rb') as f: data = f.read()\n        return len(data), len(zlib.compress(data))\n    except Exception:\n        return 0, 0\n\ndef process_single_sample(sample_id, folder_path):\n    \"\"\"Trích xuất Meta-features và Assembly features\"\"\"\n    asm_path = os.path.join(folder_path, f\"{sample_id}.asm\")\n    bytes_path = os.path.join(folder_path, f\"{sample_id}.bytes\")\n    features = {'Id': sample_id}\n    \n    bytes_raw, bytes_comp = extract_file_properties(bytes_path)\n    asm_raw, asm_comp = extract_file_properties(asm_path)\n    \n    features['bytes_raw_size'] = bytes_raw\n    features['asm_raw_size'] = asm_raw\n    features['ab_ratio'] = asm_raw / (bytes_raw + 1)\n    \n    bytes_ratio = bytes_comp / (bytes_raw + 1)\n    asm_ratio = asm_comp / (asm_raw + 1)\n    features['abc_ratio'] = asm_ratio / (bytes_ratio + 1e-5)\n    \n    section_counts = {s: 0 for s in SECTIONS}\n    opcode_counts = {op: 0 for op in OPCODES}\n    interp_counts = {char: 0 for char in INTERPUNCTIONS}\n    \n    if os.path.exists(asm_path):\n        with open(asm_path, 'r', encoding='latin-1', errors='ignore') as f:\n            for line in f:\n                line_lower = line.lower()\n                for sec in SECTIONS:\n                    if line_lower.startswith(sec.lower()): section_counts[sec] += 1\n                for op in OPCODES:\n                    if f\" {op} \" in line_lower: opcode_counts[op] += 1\n                for char in INTERPUNCTIONS:\n                    if re.search(char, line_lower): interp_counts[char] += 1\n                        \n    features.update({f\"sec_{k}\": v for k, v in section_counts.items()})\n    features.update({f\"op_{k}\": v for k, v in opcode_counts.items()})\n    features.update({f\"char_{k}\": v for k, v in interp_counts.items()})\n\n    if os.path.exists(bytes_path):\n        with open(bytes_path, 'r', errors='ignore') as f:\n            hex_data = f.read().replace('\\n', ' ').split()\n            valid_hex = [int(h, 16) for h in hex_data if h != '??']\n            if valid_hex:\n                features['entropy_mean'] = calculate_entropy(valid_hex)\n                chunks = np.array_split(valid_hex, 4)\n                for i, chunk in enumerate(chunks):\n                    features[f'entropy_q{i+1}'] = calculate_entropy(chunk)\n            else:\n                features['entropy_mean'] = 0\n                for i in range(4): features[f'entropy_q{i+1}'] = 0\n    return features","metadata":{"execution":{"iopub.execute_input":"2026-04-08T00:56:23.70749Z","iopub.status.busy":"2026-04-08T00:56:23.707172Z","iopub.status.idle":"2026-04-08T00:56:23.723149Z","shell.execute_reply":"2026-04-08T00:56:23.722262Z"},"papermill":{"duration":0.021268,"end_time":"2026-04-08T00:56:23.725203+00:00","exception":false,"start_time":"2026-04-08T00:56:23.703935+00:00","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"23d97a9a","cell_type":"code","source":"import logging\nfrom datetime import datetime\n\n# ==========================================\n# CẤU HÌNH LOGGING\n# ==========================================\nlog_file = os.path.join(WORKING_DIR, 'data_processing.log')\nlogging.basicConfig(\n    level=logging.INFO,\n    format='%(asctime)s [%(levelname)s] %(message)s',\n    datefmt='%H:%M:%S',\n    handlers=[\n        logging.FileHandler(log_file),\n        logging.StreamHandler() # In trực tiếp ra console\n    ]\n)\n\n# Bộ băm (Hasher)\nVECTORIZER = HashingVectorizer(n_features=20000, ngram_range=(2, 4), lowercase=True)\n\ndef process_batch(batch_ids, archive_path, temp_dir, batch_idx, total_batches):\n    \"\"\"Xử lý toàn diện 1 batch có kèm tracking log\"\"\"\n    \n    # 1. Giải nén\n    logging.info(f\"[Batch {batch_idx}/{total_batches}] Bắt đầu giải nén {len(batch_ids)} files...\")\n    list_file_path = os.path.join(WORKING_DIR, 'batch_list.txt')\n    with open(list_file_path, 'w') as f:\n        for m_id in batch_ids:\n            f.write(f\"{m_id}.bytes\\n\")\n            f.write(f\"{m_id}.asm\\n\")\n            \n    cmd = f\"7z e {archive_path} -o{temp_dir} -ir@{list_file_path} -y\"\n    subprocess.run(cmd, shell=True, stdout=subprocess.DEVNULL, stderr=subprocess.PIPE)\n    \n    # 2. Trích xuất Tabular Features\n    logging.info(f\"[Batch {batch_idx}/{total_batches}] Đang trích xuất Tabular features (Multiprocessing)...\")\n    args = [(uid, temp_dir) for uid in batch_ids]\n    with Pool(4) as p:\n        tabular_results = p.starmap(process_single_sample, args)\n        \n    # 3. Trích xuất N-grams\n    logging.info(f\"[Batch {batch_idx}/{total_batches}] Đang băm N-grams thành Sparse Matrix...\")\n    corpus = []\n    for uid in batch_ids:\n        bytes_path = os.path.join(temp_dir, f\"{uid}.bytes\")\n        if os.path.exists(bytes_path):\n            with open(bytes_path, 'r', errors='ignore') as f:\n                hex_content = ' '.join([w for w in f.read().split() if len(w) == 2 and w != '??'])\n                corpus.append(hex_content)\n        else:\n            corpus.append(\"\")\n            \n    ngram_sparse = VECTORIZER.transform(corpus)\n    \n    # 4. Dọn dẹp\n    logging.info(f\"[Batch {batch_idx}/{total_batches}] Hoàn tất. Đang dọn dẹp file rác giải phóng RAM...\")\n    for m_id in batch_ids:\n        b_path = os.path.join(temp_dir, f\"{m_id}.bytes\")\n        a_path = os.path.join(temp_dir, f\"{m_id}.asm\")\n        if os.path.exists(b_path): os.remove(b_path)\n        if os.path.exists(a_path): os.remove(a_path)\n        \n    gc.collect()\n    logging.info(f\"-\" * 50) # Dòng kẻ phân cách giữa các batch\n    return pd.DataFrame(tabular_results), ngram_sparse\n\ndef run_pipeline_for_dataset(ids_list, archive_path, phase_name):\n    all_tabular = []\n    all_ngrams = []\n    \n    total_batches = (len(ids_list) // BATCH_SIZE) + (1 if len(ids_list) % BATCH_SIZE != 0 else 0)\n    logging.info(f\"=== BẮT ĐẦU XỬ LÝ {phase_name.upper()} ({len(ids_list)} mẫu, {total_batches} batches) ===\")\n    \n    for i in range(0, len(ids_list), BATCH_SIZE):\n        batch_ids = ids_list[i:i+BATCH_SIZE]\n        batch_idx = (i // BATCH_SIZE) + 1\n        \n        df_tab, sparse_ng = process_batch(batch_ids, archive_path, TEMP_DIR, batch_idx, total_batches)\n        \n        all_tabular.append(df_tab)\n        all_ngrams.append(sparse_ng)\n        \n    logging.info(f\"=== ĐANG GHÉP NỐI TOÀN BỘ DỮ LIỆU {phase_name.upper()} ===\")\n    final_tabular = pd.concat(all_tabular, ignore_index=True)\n    final_ngrams = sp.vstack(all_ngrams) \n    return final_tabular, final_ngrams","metadata":{"execution":{"iopub.execute_input":"2026-04-08T00:56:23.730576Z","iopub.status.busy":"2026-04-08T00:56:23.730231Z","iopub.status.idle":"2026-04-08T00:56:23.824626Z","shell.execute_reply":"2026-04-08T00:56:23.823904Z"},"papermill":{"duration":0.099387,"end_time":"2026-04-08T00:56:23.826429+00:00","exception":false,"start_time":"2026-04-08T00:56:23.727042+00:00","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"6629b02a","cell_type":"code","source":"import math\nimport scipy.sparse as sp\n\n# =========================================================\n# CÔNG TẮC ĐIỀU KHIỂN (Hãy thay đổi biến này ở mỗi Notebook)\n# =========================================================\nTARGET = \"TEST\"   # Chọn \"TRAIN\" hoặc \"TEST\"\nTOTAL_PARTS = 8   # Tổng số phần bạn muốn chia (VD: Train = 2, Test = 6)\nPART = 4          # Khai báo phần đang chạy (Từ 1 đến TOTAL_PARTS)\n# =========================================================\n\n# 1. Đọc đúng file dữ liệu dựa trên TARGET\nif TARGET == \"TRAIN\":\n    if not os.path.exists(LABEL_CSV):\n        raise FileNotFoundError(f\"Không tìm thấy {LABEL_CSV}. Kiểm tra lại Input Kaggle!\")\n    df_labels = pd.read_csv(LABEL_CSV)\n    full_ids = df_labels['Id'].tolist()\n    y_train_full = df_labels['Class'].values - 1 # Đưa về 0-8\n    archive_path = TRAIN_7Z\n    \nelif TARGET == \"TEST\":\n    if not os.path.exists(SUBMISSION_CSV):\n        raise FileNotFoundError(f\"Không tìm thấy {SUBMISSION_CSV}. Kiểm tra lại Input Kaggle!\")\n    df_sub = pd.read_csv(SUBMISSION_CSV)\n    full_ids = df_sub['Id'].tolist()\n    archive_path = TEST_7Z\nelse:\n    raise ValueError(\"TARGET chỉ được phép là 'TRAIN' hoặc 'TEST'\")\n\n# 2. Thuật toán chia danh sách ID linh hoạt thành n phần\nchunk_size = math.ceil(len(full_ids) / TOTAL_PARTS)\nstart_idx = (PART - 1) * chunk_size\nend_idx = min(start_idx + chunk_size, len(full_ids))\n\ntarget_ids = full_ids[start_idx:end_idx]\n\n# Lấy nhãn tương ứng nếu là tập Train\nif TARGET == \"TRAIN\": \n    y_target = y_train_full[start_idx:end_idx]\n\nlogging.info(f\"🚀 BẮT ĐẦU XỬ LÝ {TARGET} - PHẦN {PART}/{TOTAL_PARTS} (Từ index {start_idx} đến {end_idx - 1} | {len(target_ids)} mẫu)\")\n\n# 3. Chạy Pipeline \ndf_data, X_ng_sparse = run_pipeline_for_dataset(target_ids, archive_path, f\"{TARGET} Part {PART}\")\n\n# 4. Xử lý Tabular Features (Bỏ cột Id)\nX_tab = df_data.drop(columns=['Id']).fillna(0).values\n\n# 5. Lưu trữ kết quả\nlogging.info(\"Đang lưu dữ liệu ra bộ nhớ đệm...\")\n\ntab_filename = os.path.join(OUT_DIR, f'X_{TARGET.lower()}_tab_part{PART}.npy')\nng_filename = os.path.join(OUT_DIR, f'X_{TARGET.lower()}_ng_part{PART}.npz')\n\nnp.save(tab_filename, X_tab)\nsp.save_npz(ng_filename, X_ng_sparse)\n\n# Chốt lưu thêm y_train nếu đang chạy tập TRAIN\nif TARGET == \"TRAIN\":\n    np.save(os.path.join(OUT_DIR, f'y_train_part{PART}.npy'), y_target)\n\nlogging.info(f\"✅ HOÀN TẤT! Dữ liệu đã lưu thành công:\\n- {tab_filename}\\n- {ng_filename}\")","metadata":{"execution":{"iopub.execute_input":"2026-04-08T00:56:23.831807Z","iopub.status.busy":"2026-04-08T00:56:23.831447Z","iopub.status.idle":"2026-04-08T09:02:53.800897Z","shell.execute_reply":"2026-04-08T09:02:53.799508Z"},"papermill":{"duration":29189.987056,"end_time":"2026-04-08T09:02:53.815384+00:00","exception":false,"start_time":"2026-04-08T00:56:23.828328+00:00","status":"completed"},"tags":[]},"outputs":[],"execution_count":null}]}