{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceType":"competition","sourceId":4117,"databundleVersionId":46665}],"dockerImageVersionId":31328,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport pandas as pd\nimport numpy as np\nimport subprocess\nimport gc\nimport xgboost as xgb\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import log_loss\nfrom sklearn.preprocessing import LabelEncoder\n\n# --- CẤU HÌNH ---\nARCHIVE_PATH = '/kaggle/input/competitions/malware-classification/train.7z'\nTEMP_DIR = '/kaggle/working/temp_extract'\nBATCH_SIZE = 100 # Tăng lên 100 file mỗi lần quét cho nhanh\nTARGET_OPCODES = ['mov', 'push', 'pop', 'jmp', 'call', 'ret', 'cmp', 'add', 'sub', 'xor', 'inc', 'dec', 'lea']\n\nos.makedirs(TEMP_DIR, exist_ok=True)\n\nlabels_df = pd.read_csv('/kaggle/input/competitions/malware-classification/trainLabels.csv')\nfile_ids = labels_df['Id'].tolist()\nall_data = []\n\nprint(f\"🚀 Bắt đầu trích xuất {len(file_ids)} file. Dự kiến 3-5 tiếng...\")\n\nfor i in range(0, len(file_ids), BATCH_SIZE):\n    batch = file_ids[i : i + BATCH_SIZE]\n    \n    # CHIẾN THUẬT: Giải nén cả cụm 100 file\n    # Dùng list comprehension để tạo danh sách file kèm đường dẫn tương đối\n    files_to_extract = []\n    for f_id in batch:\n        files_to_extract.append(f\"{f_id}.bytes\")\n        files_to_extract.append(f\"{f_id}.asm\")\n        \n    # Gọi 7z một lần cho cả batch - Đây là lý do bạn bè bạn chạy nhanh\n    cmd = ['7z', 'e', ARCHIVE_PATH] + files_to_extract + ['-o' + TEMP_DIR, '-y', '-r']\n    subprocess.run(cmd, stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL)\n    \n    # Đọc dữ liệu từ các file đã bung\n    for f_id in batch:\n        byte_path = os.path.join(TEMP_DIR, f\"{f_id}.bytes\")\n        asm_path = os.path.join(TEMP_DIR, f\"{f_id}.asm\")\n        \n        # Mặc định giá trị 0 nếu không tìm thấy file\n        byte_counts = np.zeros(257, dtype=int)\n        opcode_counts = {op: 0 for op in TARGET_OPCODES}\n        s_bytes, s_asm = 0, 0\n        \n        if os.path.exists(byte_path):\n            s_bytes = os.path.getsize(byte_path)\n            with open(byte_path, 'r') as f:\n                for line in f:\n                    tokens = line.strip().split()\n                    for hx in tokens[1:]:\n                        if hx != '??': \n                            try: byte_counts[int(hx, 16)] += 1\n                            except: pass\n                        else: byte_counts[256] += 1\n            os.remove(byte_path)\n            \n        if os.path.exists(asm_path):\n            s_asm = os.path.getsize(asm_path)\n            with open(asm_path, 'r', encoding='latin-1') as f:\n                for line in f:\n                    tokens = line.strip().split()\n                    for t in tokens:\n                        t_low = t.lower()\n                        if t_low in opcode_counts: opcode_counts[t_low] += 1\n            os.remove(asm_path)\n            \n        all_data.append(list(byte_counts) + list(opcode_counts.values()) + [s_bytes, s_asm])\n    \n    # Dọn rác RAM\n    gc.collect()\n    \n    if (i + BATCH_SIZE) % 500 == 0:\n        print(f\"✅ Đã xong {i + BATCH_SIZE} file...\")\n\n# --- LUYỆN MÔ HÌNH (Chạy rất nhanh sau khi đã có data) ---\ncolumns = [format(j, '02x').upper() for j in range(256)] + ['??'] + TARGET_OPCODES + ['size_bytes', 'size_asm']\ndf_features = pd.DataFrame(all_data, columns=columns)\ndf_features.insert(0, 'Id', file_ids)\ndf_features.to_csv('features_full.csv', index=False)\n\n# Merge và Train\nfinal_df = pd.merge(df_features, labels_df, on='Id')\nX = final_df.drop(columns=['Id', 'Class'])\ny = LabelEncoder().fit_transform(final_df['Class'])\n\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)\nmodel = xgb.XGBClassifier(tree_method='hist', n_estimators=200, max_depth=6, learning_rate=0.05)\nmodel.fit(X_train, y_train)\n\nprint(f\"\\n🎉 HOÀN TẤT! Log Loss: {log_loss(y_test, model.predict_proba(X_test)):.4f}\")\n","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}