{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceType":"competition","sourceId":4117,"databundleVersionId":46665},{"sourceType":"kernelVersion","sourceId":311677283}],"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# MS Malware Classification: Phase 3 - Mass Inference & Submission\n\n**Author:** Nguyen Le Minh Quan, Phan Ngoc Thuc, Tong Phuc Thien\n\n**Objective:** Generate predictions for the Test set. \n**Strategy:** To survive Kaggle's 12-hour limit and 30GB RAM cap, we chunk the 10,868 test files into 4 Parts. Change `PART_TO_RUN` to process a specific chunk. This script enforces strict consistency with the Adaptive CNN and Tabular features extracted in Phase 1 & 2.","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19"}},{"cell_type":"markdown","source":"## 1. Environment Setup & Part Configuration","metadata":{}},{"cell_type":"code","source":"import os\nimport gc\nimport subprocess\nimport cv2\nimport numpy as np\nimport pandas as pd\nimport joblib\nimport xgboost as xgb\nimport torch\nimport torch.nn as nn\nfrom torchvision import transforms\nfrom tqdm.notebook import tqdm\nimport warnings\nimport logging\n\nwarnings.filterwarnings('ignore')\nlogging.getLogger(\"xgboost\").setLevel(logging.ERROR)\n\n# --- 1. CHUNK CONFIGURATION ---\nPART_TO_RUN = 4\n\nPHASE2_DIR = \"/kaggle/input/notebooks/nguynlminhqun/ms-mal-classification-hybrid-architecture\" \nARCHIVE_PATH = \"/kaggle/input/competitions/malware-classification/test.7z\"\nSUBMISSION_TEMPLATE = \"/kaggle/input/competitions/malware-classification/sampleSubmission.csv\"\n\nBATCH_DIR = \"/dev/shm/test_batch\" \nos.makedirs(BATCH_DIR, exist_ok=True)\n\n# Parse Test IDs and split into 4 parts\ndf_sub = pd.read_csv(SUBMISSION_TEMPLATE)\ntest_ids_full = df_sub['Id'].tolist()\n\nchunk_size = len(test_ids_full) // 4\nchunks = [\n    test_ids_full[0 : chunk_size],                             # Part 1: 0 -> ~2717\n    test_ids_full[chunk_size : chunk_size*2],                  # Part 2: ~2717 -> ~5434\n    test_ids_full[chunk_size*2 : chunk_size*3],                # Part 3: ~5434 -> ~8151\n    test_ids_full[chunk_size*3 : ]                             # Part 4: ~8151 -> End\n]\n\ntest_ids = chunks[PART_TO_RUN - 1]\nprint(f\"--- RUNNING PART {PART_TO_RUN} ---\")\nprint(f\"Total files to predict in this chunk: {len(test_ids)}\")\n\n# Constants for Feature Extraction (MUST MATCH PHASE 1 EXACTLY)\nHEX_CODES = [f\"{i:02x}\".upper() for i in range(256)]\nSEGMENTS = ['.text', '.data', '.bss', '.rdata', '.edata', '.idata', '.rsrc']\nOPCODES = ['mov', 'push', 'pop', 'jmp', 'call', 'ret', 'cmp', 'test', 'add', 'sub', 'inc', 'dec', 'xor', 'jz', 'jnz']","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 2. Load Phase 2 Artifacts (Adaptive CNN, Final XGBoost, Final Scaler)","metadata":{}},{"cell_type":"code","source":"device = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\nprint(f\"Inference Device: {device}\")\n\n# 1. Load Scaler\nscaler = joblib.load(os.path.join(PHASE2_DIR, 'tabular_scaler_final.pkl'))\n\n# 2. Rebuild Adaptive CNN\nclass AdaptiveMalwareCNN(nn.Module):\n    def __init__(self, num_classes=9):\n        super(AdaptiveMalwareCNN, self).__init__()\n        self.features = nn.Sequential(\n            nn.Conv2d(1, 32, kernel_size=3, padding=1), nn.ReLU(), nn.MaxPool2d(2, 2),\n            nn.Conv2d(32, 64, kernel_size=3, padding=1), nn.ReLU(), nn.MaxPool2d(2, 2),\n            nn.Conv2d(64, 128, kernel_size=3, padding=1), nn.ReLU(), nn.MaxPool2d(2, 2)\n        )\n        self.adaptive_pool = nn.AdaptiveAvgPool2d((4, 4)) \n        self.flatten_size = 128 * 4 * 4\n        self.classifier = nn.Sequential(\n            nn.Linear(self.flatten_size, 256),\n            nn.ReLU(),\n            nn.Dropout(0.5),\n            nn.Linear(256, num_classes)\n        )\n\n    def extract_features(self, x):\n        x = self.features(x)\n        x = self.adaptive_pool(x)\n        x = x.view(x.size(0), -1)\n        x = self.classifier[0](x) \n        x = self.classifier[1](x) \n        return x\n\ncnn_model = AdaptiveMalwareCNN(num_classes=9).to(device)\ncnn_model.load_state_dict(torch.load(os.path.join(PHASE2_DIR, 'adaptive_cnn_weights.pth'), map_location=device))\ncnn_model.eval()\n\n# 3. Load XGBoost\nxgb_model = xgb.XGBClassifier()\nxgb_model.load_model(os.path.join(PHASE2_DIR, 'xgboost_model_final.json'))\n\nimg_transform = transforms.ToTensor()\nprint(\"All Models and Final Scaler loaded successfully.\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 3. Streaming Inference Engine (CPU + GPU Sync)\nExtract files to `/dev/shm`, extract Tabular features + Opcodes, render image array, run through Adaptive CNN with `batch_size=1` (due to variable dimensions), merge, predict, and purge RAM.","metadata":{}},{"cell_type":"code","source":"def get_image_width(size_kb):\n    if size_kb < 10: return 64\n    if size_kb < 30: return 128\n    if size_kb < 60: return 256\n    if size_kb < 100: return 384\n    if size_kb < 200: return 512\n    if size_kb < 500: return 768\n    if size_kb < 1000: return 1024\n    return 2048\n\ndef process_and_predict_batch(batch_ids):\n    # 1. Extract to RAM Disk\n    with open('batch_list.txt', 'w') as f:\n        for fid in batch_ids:\n            f.write(f\"test/{fid}.bytes\\n\")\n            f.write(f\"test/{fid}.asm\\n\")\n            \n    subprocess.run(['7z', 'e', ARCHIVE_PATH, '@batch_list.txt', f'-o{BATCH_DIR}', '-y'], stdout=subprocess.DEVNULL)\n    \n    batch_cnn_features, batch_tabular = [], []\n    \n    # 2. Process Files\n    for fid in batch_ids:\n        byte_path = os.path.join(BATCH_DIR, f\"{fid}.bytes\")\n        asm_path = os.path.join(BATCH_DIR, f\"{fid}.asm\")\n        \n        tab_vector = []\n        img_tensor = None\n        \n        # Read .bytes file\n        if os.path.exists(byte_path):\n            size_kb = os.path.getsize(byte_path) / 1024\n            width = get_image_width(size_kb)\n            \n            with open(byte_path, 'r') as f:\n                content = f.read()\n                \n                # 1. Unknown bytes feature\n                tab_vector.append(content.count('??'))\n                \n                # Clean and split\n                tokens = content.replace('??', '00').split()\n            \n            hex_iter = (int(t, 16) for t in tokens if len(t) == 2)\n            int_data = np.fromiter(hex_iter, dtype=np.uint8)\n            \n            # 2. Byte Frequencies\n            byte_counts = np.bincount(int_data, minlength=256)\n            tab_vector.extend(byte_counts)\n            \n            # Generate Image Array (NO RESIZE)\n            height = len(int_data) // width\n            img_array = int_data[:height * width].reshape((height, width))\n            img_tensor = img_transform(img_array).unsqueeze(0).to(device) # Shape: [1, 1, H, W]\n            \n            os.remove(byte_path)\n        else:\n            # Fallback for missing .bytes\n            tab_vector.append(0)\n            tab_vector.extend([0]*256)\n            img_tensor = torch.zeros((1, 1, 224, 224), dtype=torch.float32).to(device)\n\n        # Read .asm file\n        if os.path.exists(asm_path):\n            with open(asm_path, 'r', encoding='latin1') as f:\n                asm_content = f.read().lower()\n                \n            # 3. ASM Size\n            tab_vector.append(len(asm_content))\n            # 4. Segments\n            for seg in SEGMENTS: \n                tab_vector.append(asm_content.count(seg + ':'))\n            # 5. Opcodes\n            for op in OPCODES: \n                tab_vector.append(asm_content.count(f' {op} '))\n                \n            os.remove(asm_path)\n        else:\n            # Fallback for missing .asm\n            tab_vector.extend([0] * (1 + len(SEGMENTS) + len(OPCODES)))\n            \n        # Ensure correct column matching!\n        batch_tabular.append(tab_vector)\n        \n        # Inference CNN individually due to variable aspect ratios\n        with torch.no_grad():\n            cnn_feat = cnn_model.extract_features(img_tensor).cpu().numpy()[0]\n            batch_cnn_features.append(cnn_feat)\n    \n    # 3. Scale and Predict using Hybrid Model\n    X_hybrid_raw = np.hstack((np.array(batch_cnn_features), np.array(batch_tabular)))\n    X_hybrid_scaled = scaler.transform(X_hybrid_raw) \n    batch_probs = xgb_model.predict_proba(X_hybrid_scaled)\n    \n    if os.path.exists('batch_list.txt'): os.remove('batch_list.txt')\n    return batch_probs","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 4. Batch Execution & Final Export","metadata":{}},{"cell_type":"code","source":"BATCH_SIZE = 150\nall_probabilities = []\n\nprint(f\"Starting batch prediction process for Part {PART_TO_RUN}...\")\n\nfor i in tqdm(range(0, len(test_ids), BATCH_SIZE), desc=f\"Processing Batches (Part {PART_TO_RUN})\"):\n    batch_ids = test_ids[i : i + BATCH_SIZE]\n    probs = process_and_predict_batch(batch_ids)\n    all_probabilities.extend(probs)\n    \n    gc.collect()\n    torch.cuda.empty_cache()\n\n# Generate Kaggle-ready format\nclass_columns = [f'Prediction{i}' for i in range(1, 10)]\ndf_final = pd.DataFrame(all_probabilities, columns=class_columns)\ndf_final.insert(0, 'Id', test_ids)\ndf_final.to_csv(f\"sub_part{PART_TO_RUN}.csv\", index=False)\n\nprint(f\"Process complete. sub_part{PART_TO_RUN}.csv generated successfully.\")","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}