{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":4117,"databundleVersionId":46665,"sourceType":"competition"},{"sourceId":14401390,"sourceType":"datasetVersion","datasetId":9197700}],"dockerImageVersionId":31234,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport glob\nimport numpy as np\nimport pandas as pd\nimport joblib\nimport xgboost as xgb\nfrom sklearn.feature_extraction.text import CountVectorizer\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import accuracy_score, classification_report, confusion_matrix\n\n# ===================== CONFIG =====================\n# Đường dẫn dataset gốc (chứa trainLabels.csv) - Bạn kiểm tra lại đúng tên dataset trên Kaggle nhé\n# Thường là: /kaggle/input/malware-classification/\nBASE_DATA_DIR = \"/kaggle/input/malware-classification\" \nLABELS_CSV = os.path.join(BASE_DATA_DIR, \"trainLabels.csv\")\n\n# Đường dẫn dataset feature đã xử lý (từ ảnh bạn gửi)\n# Lưu ý: Kaggle mount dataset vào /kaggle/input/\nRF_GINI_DIR = \"/kaggle/input/rf-gini-out\"\nVOCAB_FILE = os.path.join(RF_GINI_DIR, \"rf_gini_out/final_vocab_rf_ranked.txt\")\n\n# Lấy tất cả các file list opcode\nLIST_FILES = sorted(glob.glob(os.path.join(RF_GINI_DIR, \"listfile_RF_*.txt\")))\n\n# Config Model (Tối ưu chống Overfit)\nSEED = 20251226\nMAX_FEATURES = 2000 # Chỉ lấy top 2000 features quan trọng nhất từ file vocab để nhẹ máy và tránh nhiễu\nTEST_SIZE = 0.2\n\nXGB_PARAMS = {\n    'objective': 'multi:softprob',\n    'num_class': 9,\n    'tree_method': 'hist',      # Tăng tốc độ train\n    'eval_metric': 'mlogloss',\n    'seed': SEED,\n    'n_jobs': -1,\n    \n    # --- CHỐNG OVERFITTING CONFIG ---\n    'eta': 0.05,                # Learning rate thấp (0.05 thay vì 0.3)\n    'max_depth': 6,             # Độ sâu cây vừa phải (tránh học vẹt)\n    'min_child_weight': 3,      # Tăng lên để tránh chia node cho các nhóm mẫu quá nhỏ\n    'subsample': 0.7,           # Chỉ dùng 70% dữ liệu mỗi lần dựng cây\n    'colsample_bytree': 0.6,    # Chỉ dùng 60% features mỗi lần dựng cây\n    'reg_alpha': 0.1,           # L1 Regularization\n    'reg_lambda': 1.5,          # L2 Regularization\n}","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def load_vocab(vocab_path, top_k=None):\n    \"\"\"Load vocabulary từ file txt, mỗi dòng là 1 feature\"\"\"\n    print(f\"Loading vocab from: {vocab_path}\")\n    with open(vocab_path, 'r', encoding='utf-8') as f:\n        # Giả sử format file là: \"feature_name score\" hoặc chỉ \"feature_name\"\n        # Code này xử lý trường hợp chỉ có tên feature trên mỗi dòng\n        words = [line.strip().split()[0] for line in f if line.strip()]\n    \n    if top_k:\n        words = words[:top_k]\n    print(f\"-> Loaded {len(words)} features.\")\n    return words\n\ndef load_data_from_txt_files(file_paths):\n    \"\"\"Đọc nội dung opcode từ danh sách các file txt\"\"\"\n    docs = []\n    print(f\"Reading {len(file_paths)} list files...\")\n    for path in file_paths:\n        with open(path, 'r', encoding='utf-8', errors='ignore') as f:\n            # Giả sử mỗi file chứa nhiều dòng, mỗi dòng là 1 văn bản (opcode sequence) của 1 sample\n            # Hoặc cả file là 1 list các opcode nối liền.\n            # Dựa trên tên 'listfile', mình giả định mỗi dòng trong file này tương ứng với 1 sample ID theo thứ tự\n            content = f.readlines()\n            docs.extend([line.strip() for line in content])\n    print(f\"-> Total documents loaded: {len(docs)}\")\n    return docs\n\ndef load_labels(csv_path):\n    \"\"\"Load nhãn từ csv\"\"\"\n    df = pd.read_csv(csv_path)\n    # Sắp xếp theo Id để đảm bảo khớp thứ tự với dữ liệu đọc vào\n    # (Rất quan trọng: Dữ liệu listfile_RF thường được generate theo thứ tự sorted ID)\n    df = df.sort_values(by=\"Id\")\n    return df[\"Class\"].values - 1  # Chuyển về 0..8 cho XGBoost","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ===================== CELL 3: DATA PREPARATION =====================\nprint(\"--- STARTING DATA PREPARATION ---\")\n\n# 1. Load Vocab\nvocab = load_vocab(VOCAB_FILE, top_k=MAX_FEATURES)\n\n# 2. Load Documents (Opcodes)\n# Đảm bảo LIST_FILES tìm thấy file. Nếu list rỗng, hãy kiểm tra lại đường dẫn\nif not LIST_FILES:\n    print(f\"[ERROR] Không tìm thấy file nào trong {RF_GINI_DIR}. Kiểm tra lại đường dẫn!\")\nelse:\n    raw_docs = load_data_from_txt_files(LIST_FILES)\n\n# 3. Load Labels\n# Load labels và đảm bảo khớp số lượng\ny_all = load_labels(LABELS_CSV)\n\n# Kiểm tra độ khớp dữ liệu\nif len(raw_docs) != len(y_all):\n    print(f\"[WARNING] Mismatch: Docs={len(raw_docs)}, Labels={len(y_all)}\")\n    # Cắt ngắn label hoặc docs cho bằng nhau để code không bị crash\n    min_len = min(len(raw_docs), len(y_all))\n    raw_docs = raw_docs[:min_len]\n    y_all = y_all[:min_len]\n    print(f\"-> Adjusted both to {min_len} samples.\")\nelse:\n    print(f\"-> Data matched perfectly: {len(raw_docs)} samples.\")\n\n# 4. Vectorize\nprint(\"Vectorizing data...\")\nvectorizer = CountVectorizer(\n    vocabulary=vocab,  # CHỈ DÙNG feature có trong vocab\n    tokenizer=lambda x: x.split(), # Tokenizer đơn giản tách theo khoảng trắng\n    token_pattern=None \n)\n\nX_all = vectorizer.transform(raw_docs)\nprint(\"Shape of X matrix:\", X_all.shape)\n\n# 5. Split Train/Val\n# Đây là bước tạo ra biến X_train mà Cell 4 cần\nX_train, X_val, y_train, y_val = train_test_split(\n    X_all, y_all, test_size=TEST_SIZE, stratify=y_all, random_state=SEED\n)\n\nprint(\"\\n--- DATA PREPARATION COMPLETE ---\")\nprint(\"Train shape:\", X_train.shape)\nprint(\"Val shape:\", X_val.shape)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ===================== CELL 4: TRAINING =====================\n# Kiểm tra lại lần nữa xem X_train đã tồn tại chưa\nif 'X_train' not in locals():\n    raise ValueError(\"CHƯA CÓ DỮ LIỆU! Bạn vui lòng chạy Cell 3 trước nhé.\")\n\nprint(\"--- STARTING TRAINING ---\")\n\n# Tạo DMatrix cho XGBoost (nhanh hơn format thường)\ndtrain = xgb.DMatrix(X_train, label=y_train)\ndval = xgb.DMatrix(X_val, label=y_val)\n\n# Danh sách watchlist để theo dõi quá trình train\nwatchlist = [(dtrain, 'train'), (dval, 'val')]\n\nprint(\"Training XGBoost with Early Stopping...\")\n# Train model\nmodel = xgb.train(\n    params=XGB_PARAMS,\n    dtrain=dtrain,\n    num_boost_round=2000,           # Cho phép train tối đa 2000 vòng\n    evals=watchlist,\n    early_stopping_rounds=50,       # Dừng nếu sau 50 vòng val_loss không giảm\n    verbose_eval=50                 # In kết quả mỗi 50 vòng\n)\n\nprint(\"\\nTraining Finished.\")\nprint(f\"Best Iteration: {model.best_iteration}\")\nprint(f\"Best Score (MLogloss): {model.best_score}\")\n\n# Dự đoán\nprint(\"Evaluating on Validation set...\")\npreds_prob = model.predict(dval)\npreds_class = np.argmax(preds_prob, axis=1)\n\n# Đánh giá\nacc = accuracy_score(y_val, preds_class)\nprint(f\"\\nAccuracy: {acc:.4f}\")\nprint(\"Classification Report:\")\nprint(classification_report(y_val, preds_class, digits=4))\n\n# Confusion Matrix (optional)\nprint(\"Confusion Matrix:\")\nprint(confusion_matrix(y_val, preds_class))\n\n# Lưu model\nout_path = \"xgb_optimized_model.json\"\nmodel.save_model(out_path)\nprint(f\"Model saved to {out_path}\")","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}