{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceType":"competition","sourceId":4117,"databundleVersionId":46665}],"dockerImageVersionId":31328,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\n\n# 1. Tạo thư mục làm việc\n!rm -rf /kaggle/working/temp_bytes /kaggle/working/malware_images\n!mkdir -p /kaggle/working/temp_bytes /kaggle/working/malware_images\n\n# 2. Lấy danh sách 1000 file .bytes đầu tiên\n!7z l /kaggle/input/competitions/malware-classification/train.7z | grep .bytes | awk '{print $NF}' | head -n 1000 > file_list.txt\n\n# 3. Giải nén (Chỉ trích xuất file trong danh sách)\n!7z e /kaggle/input/competitions/malware-classification/train.7z -o/kaggle/working/temp_bytes -i@file_list.txt -y > /dev/null\n\nprint(f\"Đã chuẩn bị xong {len(os.listdir('/kaggle/working/temp_bytes'))} file thô.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-19T11:31:09.170423Z","iopub.execute_input":"2026-04-19T11:31:09.171711Z","iopub.status.idle":"2026-04-19T11:32:13.608487Z","shell.execute_reply.started":"2026-04-19T11:31:09.171622Z","shell.execute_reply":"2026-04-19T11:32:13.607161Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nfrom PIL import Image\nimport glob\nimport torch\nfrom torch.utils.data import Dataset, DataLoader, Subset\nfrom torchvision import transforms, models\nfrom sklearn.model_selection import train_test_split\n\n# Đọc nhãn\nlabels_df = pd.read_csv('/kaggle/input/competitions/malware-classification/trainLabels.csv')\nlabel_dict = dict(zip(labels_df.Id, labels_df.Class))\n\n# Hàm xử lý ảnh\ndef process_to_images():\n    files = glob.glob('/kaggle/working/temp_bytes/*.bytes')\n    for f_path in files:\n        fid = os.path.basename(f_path).replace('.bytes', '')\n        with open(f_path, 'r') as f:\n            hex_list = [int(x, 16) for x in f.read().split() if len(x) == 2 and x != '??']\n        if not hex_list: continue\n        data = np.array(hex_list, dtype=np.uint8)\n        w = int(len(data)**0.5)\n        img = Image.fromarray(data[:w*(len(data)//w)].reshape((len(data)//w, w)))\n        img = img.resize((128, 128), Image.Resampling.LANCZOS)\n        img.save(f\"/kaggle/working/malware_images/{label_dict[fid]}_{fid}.png\")\n\nprocess_to_images()\n!rm -rf /kaggle/working/temp_bytes # Xóa thô ngay để giải phóng Disk\n\n# Định nghĩa Dataset\nclass MalwareDataset(Dataset):\n    def __init__(self, img_dir, transform=None):\n        self.img_names = [f for f in os.listdir(img_dir) if f.endswith('.png')]\n        self.img_dir = img_dir\n        self.transform = transform\n    def __len__(self): return len(self.img_names)\n    def __getitem__(self, idx):\n        name = self.img_names[idx]\n        label = int(name.split('_')[0]) - 1\n        img = Image.open(os.path.join(self.img_dir, name)).convert('L')\n        if self.transform: img = self.transform(img)\n        return img, label\n\ntransform = transforms.Compose([\n    transforms.RandomHorizontalFlip(),\n    transforms.ToTensor(),\n    transforms.Normalize((0.5,), (0.5,))\n])\n\nds = MalwareDataset('/kaggle/working/malware_images', transform=transform)\ntrain_idx, val_idx = train_test_split(range(len(ds)), test_size=0.2, random_state=42)\ntrain_loader = DataLoader(Subset(ds, train_idx), batch_size=16, shuffle=True)\nval_loader = DataLoader(Subset(ds, val_idx), batch_size=16)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-19T11:32:19.562014Z","iopub.execute_input":"2026-04-19T11:32:19.562327Z","iopub.status.idle":"2026-04-19T11:36:00.980130Z","shell.execute_reply.started":"2026-04-19T11:32:19.562292Z","shell.execute_reply":"2026-04-19T11:36:00.979216Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import torch.nn as nn\n\ndevice = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\n\n# Khởi tạo ResNet18 tối ưu cho ảnh xám 9 lớp\nmodel = models.resnet18(weights=models.ResNet18_Weights.DEFAULT)\nmodel.conv1 = nn.Conv2d(1, 64, kernel_size=7, stride=2, padding=3, bias=False)\nmodel.fc = nn.Linear(model.fc.in_features, 9)\nmodel = model.to(device)\n\ncriterion = nn.CrossEntropyLoss()\noptimizer = torch.optim.Adam(model.parameters(), lr=0.0003) # lr thấp cho ResNet\n\n# Training loop\nfor epoch in range(15):\n    model.train()\n    total_loss = 0\n    for imgs, labels in train_loader:\n        imgs, labels = imgs.to(device), labels.to(device)\n        optimizer.zero_grad()\n        loss = criterion(model(imgs), labels)\n        loss.backward()\n        optimizer.step()\n        total_loss += loss.item()\n    \n    # Eval\n    model.eval()\n    correct = 0\n    with torch.no_grad():\n        for imgs, labels in val_loader:\n            outputs = model(imgs.to(device))\n            correct += (outputs.argmax(1) == labels.to(device)).sum().item()\n    \n    print(f\"Epoch {epoch+1}: Loss {total_loss/len(train_loader):.4f} | Acc {100*correct/len(val_idx):.2f}%\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-19T11:36:13.759573Z","iopub.execute_input":"2026-04-19T11:36:13.759935Z","iopub.status.idle":"2026-04-19T11:43:23.602642Z","shell.execute_reply.started":"2026-04-19T11:36:13.759901Z","shell.execute_reply":"2026-04-19T11:43:23.601198Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Đọc danh sách cũ và tạo danh sách mới cho .asm\nwith open('file_list.txt', 'r') as f:\n    asm_list = [line.strip().replace('.bytes', '.asm') for line in f.readlines()]\n\n# Lưu vào file để lệnh 7z sử dụng\nwith open('asm_file_list.txt', 'w') as f:\n    for item in asm_list:\n        f.write(f\"{item}\\n\")\n\nprint(f\"Đã chuẩn bị danh sách cho {len(asm_list)} file .asm\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-19T11:43:33.842800Z","iopub.execute_input":"2026-04-19T11:43:33.843197Z","iopub.status.idle":"2026-04-19T11:43:33.857649Z","shell.execute_reply.started":"2026-04-19T11:43:33.843172Z","shell.execute_reply":"2026-04-19T11:43:33.856367Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\n\n# Tạo thư mục chứa file .asm tạm thời\ntemp_asm_path = '/kaggle/working/temp_asm'\n!mkdir -p {temp_asm_path}\n\n# Lệnh 7z trích xuất dựa trên danh sách asm_file_list.txt\n# Lưu ý: -i@ là tham số để đọc file danh sách\nprint(\"Bắt đầu trích xuất... Vui lòng đợi trong giây lát.\")\n\n!7z e /kaggle/input/competitions/malware-classification/train.7z -o{temp_asm_path} -i@asm_file_list.txt -y > /dev/null\n\n# Kiểm tra số lượng file đã giải nén thành công\nextracted_asm = os.listdir(temp_asm_path)\nprint(f\"Thành công! Đã trích xuất {len(extracted_asm)} file .asm vào thư mục {temp_asm_path}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-19T11:43:36.880913Z","iopub.execute_input":"2026-04-19T11:43:36.881185Z","iopub.status.idle":"2026-04-19T11:45:36.951689Z","shell.execute_reply.started":"2026-04-19T11:43:36.881163Z","shell.execute_reply":"2026-04-19T11:45:36.950372Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport re\nimport pandas as pd\n\ndef fast_asm_processor(img_dir):\n    # 1. Danh sách các đặc trưng cần tìm\n    opcodes = ['jmp', 'mov', 'retf', 'push', 'pop', 'add', 'sub', 'mul', 'div', 'and', 'or', 'xor', 'cmp', 'test', 'call']\n    sections = ['.text', '.data', '.bss', '.rdata', '.edata', '.idata', '.rsrc', '.tls', '.reloc']\n    keywords = ['db', 'dw', 'dd', 'offset', 'api', '.dll', 'std::']\n    \n    asm_data = []\n    files = [f for f in os.listdir(img_dir) if f.endswith('.asm')]\n    \n    print(f\"Bắt đầu quét {len(files)} file...\")\n\n    for i, fname in enumerate(files):\n        fpath = os.path.join(img_dir, fname)\n        fid = fname.replace('.asm', '')\n        \n        # Khởi tạo dictionary chứa các đặc trưng\n        feats = {f'asm_{item}': 0 for item in (opcodes + sections + keywords)}\n        feats['Id'] = fid\n        \n        try:\n            with open(fpath, 'r', encoding='latin-1') as f:\n                for line in f:\n                    line = line.lower()\n                    # Quét nhanh qua từng dòng\n                    for op in opcodes:\n                        if op in line: feats[f'asm_{op}'] += 1\n                    for sec in sections:\n                        if sec in line: feats[f'asm_{sec}'] += 1\n                    for key in keywords:\n                        if key in line: feats[f'asm_{key}'] += 1\n        except:\n            pass\n            \n        asm_data.append(feats)\n        if (i+1) % 100 == 0:\n            print(f\"Đã xử lý {i+1}/999 file...\")\n\n    return pd.DataFrame(asm_data)\n\n# Chạy trích xuất\ndf_asm_features = fast_asm_processor('/kaggle/working/temp_asm')\n\n# Merge với nhãn (Labels)\nlabels_df = pd.read_csv('/kaggle/input/competitions/malware-classification/trainLabels.csv')\ndf_final_asm = pd.merge(df_asm_features, labels_df, on='Id')\n\n# Lưu lại file CSV và XÓA NGAY thư mục temp_asm để tránh đầy bộ nhớ\ndf_final_asm.to_csv('asm_features_final.csv', index=False)\n!rm -rf /kaggle/working/temp_asm\n\nprint(\"\\nĐã lưu đặc trưng vào 'asm_features_final.csv' và dọn dẹp bộ nhớ!\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-19T11:46:10.036266Z","iopub.execute_input":"2026-04-19T11:46:10.037603Z","iopub.status.idle":"2026-04-19T11:54:25.899187Z","shell.execute_reply.started":"2026-04-19T11:46:10.037541Z","shell.execute_reply":"2026-04-19T11:54:25.897693Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from xgboost import XGBClassifier\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import accuracy_score, classification_report\n\n# Chuẩn bị dữ liệu\nX = df_final_asm.drop(['Id', 'Class'], axis=1)\ny = df_final_asm['Class'] - 1 # Chuyển nhãn về 0-8\n\nX_train, X_val, y_train, y_val = train_test_split(X, y, test_size=0.2, random_state=42)\n\n# Khởi tạo và huấn luyện XGBoost\nxgb_model = XGBClassifier(\n    n_estimators=300,\n    max_depth=5,\n    learning_rate=0.1,\n    tree_method='gpu_hist' if torch.cuda.is_available() else 'auto', # Dùng GPU nếu có\n    random_state=42\n)\n\nxgb_model.fit(X_train, y_train)\n\n# Đánh giá\ny_pred = xgb_model.predict(X_val)\nprint(f\"Độ chính xác XGBoost (ASM): {accuracy_score(y_val, y_pred)*100:.2f}%\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-19T11:54:54.782998Z","iopub.execute_input":"2026-04-19T11:54:54.783350Z","iopub.status.idle":"2026-04-19T11:54:55.768677Z","shell.execute_reply.started":"2026-04-19T11:54:54.783317Z","shell.execute_reply":"2026-04-19T11:54:55.767962Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\n\ndef final_ensemble_predict(img_tensor, asm_numeric_features):\n    # 1. Dự đoán từ ResNet (Ảnh)\n    model.eval()\n    with torch.no_grad():\n        img_output = model(img_tensor.to(device))\n        # Lấy xác suất sau Softmax\n        prob_resnet = torch.softmax(img_output, dim=1).cpu().numpy()\n    \n    # 2. Dự đoán từ XGBoost (ASM)\n    # Lưu ý: asm_numeric_features cần là dạng bảng (DataFrame hoặc array)\n    prob_xgb = xgb_model.predict_proba(asm_numeric_features)\n    \n    # 3. Kết hợp (Trọng số 50-50 hoặc tùy chỉnh)\n    # final_prob = (Trọng số * ResNet) + (Trọng số * XGBoost)\n    final_prob = (0.4 * prob_resnet) + (0.6 * prob_xgb)\n    \n    # Lấy lớp có xác suất cao nhất\n    final_class = np.argmax(final_prob, axis=1)\n    return final_class","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-19T11:55:42.469142Z","iopub.execute_input":"2026-04-19T11:55:42.469512Z","iopub.status.idle":"2026-04-19T11:55:42.476179Z","shell.execute_reply.started":"2026-04-19T11:55:42.469482Z","shell.execute_reply":"2026-04-19T11:55:42.475196Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Lấy dự đoán từ tập Validation của cả 2\n# (Giả sử bạn đã có y_pred_resnet và y_pred_xgb cho cùng một tập file)\n\nfrom sklearn.metrics import confusion_matrix\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\n# Vẽ Heatmap để xem mô hình hay nhầm lẫn ở loại malware nào nhất\ncm = confusion_matrix(y_val, y_pred) # y_pred của XGBoost\nplt.figure(figsize=(10,8))\nsns.heatmap(cm, annot=True, fmt='d', cmap='Blues')\nplt.xlabel('Dự đoán')\nplt.ylabel('Thực tế')\nplt.title('Ma trận nhầm lẫn - ASM XGBoost')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-19T11:55:54.025099Z","iopub.execute_input":"2026-04-19T11:55:54.025503Z","iopub.status.idle":"2026-04-19T11:55:54.776859Z","shell.execute_reply.started":"2026-04-19T11:55:54.025461Z","shell.execute_reply":"2026-04-19T11:55:54.775920Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\n\ndef create_submission(resnet_probs, xgb_probs, file_ids):\n    # Kết hợp xác suất (Trọng số 0.4 cho Ảnh, 0.6 cho ASM dựa trên Acc 98%)\n    final_probs = (0.4 * resnet_probs) + (0.6 * xgb_probs)\n    \n    # Tạo DataFrame theo đúng format Kaggle\n    columns = ['Prediction1', 'Prediction2', 'Prediction3', 'Prediction4', \n               'Prediction5', 'Prediction6', 'Prediction7', 'Prediction8', 'Prediction9']\n    \n    submission = pd.DataFrame(final_probs, columns=columns)\n    submission.insert(0, 'Id', file_ids)\n    \n    # Lưu file\n    submission.to_csv('submission.csv', index=False)\n    print(\"Đã tạo file submission.csv thành công!\")\n\n# Giả sử bạn đã có:\n# - resnet_probs: Ma trận (999, 9) từ model.predict()\n# - xgb_probs: Ma trận (999, 9) từ xgb_model.predict_proba()\n# - val_file_ids: Danh sách ID của các file trong tập Validation","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-19T11:59:41.318907Z","iopub.execute_input":"2026-04-19T11:59:41.319250Z","iopub.status.idle":"2026-04-19T11:59:41.326926Z","shell.execute_reply.started":"2026-04-19T11:59:41.319219Z","shell.execute_reply":"2026-04-19T11:59:41.325665Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Lưu mô hình ResNet\ntorch.save(model.state_dict(), 'malware_resnet.pth')\n\n# Lưu mô hình XGBoost\nxgb_model.save_model('malware_xgb.json')\n\nprint(\"Đã đóng gói bộ đôi model hoàn chỉnh!\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-19T11:57:07.735813Z","iopub.execute_input":"2026-04-19T11:57:07.736090Z","iopub.status.idle":"2026-04-19T11:57:07.876161Z","shell.execute_reply.started":"2026-04-19T11:57:07.736069Z","shell.execute_reply":"2026-04-19T11:57:07.875244Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import torch\nimport torch.nn.functional as F\n\n# 1. Lấy xác suất từ ResNet (Ảnh)\nmodel.eval()\nresnet_probs_list = []\nids_list = []\n\nwith torch.no_grad():\n    # Chúng ta dùng luôn val_loader đã tạo ở các bước trước\n    for images, labels in val_loader: \n        outputs = model(images.to(device))\n        probs = F.softmax(outputs, dim=1)\n        resnet_probs_list.append(probs.cpu().numpy())\n\nresnet_probs = np.vstack(resnet_probs_list)\n\n# 2. Lấy xác suất từ XGBoost (ASM)\n# X_val là tập đặc trưng ASM bạn đã chia ở bước XGBoost\nxgb_probs = xgb_model.predict_proba(X_val)\n\n# 3. Lấy ID tương ứng (Giả sử bạn lấy từ tập val_idx)\n# Lưu ý: Đảm bảo thứ tự file trong ResNet và XGBoost phải khớp nhau\nval_file_ids = [ds.img_names[i].split('_')[1].replace('.png','') for i in val_idx]\n\nprint(\"Đã chuẩn bị xong dữ liệu dự đoán!\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-19T12:01:27.532184Z","iopub.execute_input":"2026-04-19T12:01:27.532501Z","iopub.status.idle":"2026-04-19T12:01:29.695844Z","shell.execute_reply.started":"2026-04-19T12:01:27.532477Z","shell.execute_reply":"2026-04-19T12:01:29.695109Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Giờ thì các biến đã tồn tại, không còn lỗi NameError nữa\ncreate_submission(resnet_probs, xgb_probs, val_file_ids)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-19T12:01:39.716601Z","iopub.execute_input":"2026-04-19T12:01:39.716932Z","iopub.status.idle":"2026-04-19T12:01:39.727368Z","shell.execute_reply.started":"2026-04-19T12:01:39.716902Z","shell.execute_reply":"2026-04-19T12:01:39.726626Z"}},"outputs":[],"execution_count":null}]}