{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"},"kaggle":{"accelerator":"none","dataSources":[{"sourceType":"competition","sourceId":4117,"databundleVersionId":46665,"isSourceIdPinned":false}],"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport re\nimport gc\nimport time\nimport json\nimport subprocess\nimport numpy as np\nimport pandas as pd\nimport multiprocessing as mp\nfrom kaggle_secrets import UserSecretsClient\nfrom kaggle.api.kaggle_api_extended import KaggleApi\n\n# ==============================================================================\n# CẤU HÌNH CHIẾN THUẬT - TỐI ƯU HÓA CHO TẬP TEST\n# ==============================================================================\nTOTAL_PARTS = 4          # Đã trả về 4 phần để nhờ bạn bè chạy cùng\nMY_PART = 0              # ĐỔI THÀNH: 0, 1, 2, 3 tùy máy người chạy\nBATCH_SIZE = 400         # TĂNG LÊN 400 để giảm thiểu số lần 7z phải quét lại file\nTIMEOUT_LIMIT = 11.5 * 3600 \nKAGGLE_USER = \"padduwcs\" # Tên username Kaggle của bạn\n\nSTART_TIME = time.time()\nTEST_SUBMISSION_CSV = '/kaggle/input/competitions/malware-classification/sampleSubmission.csv'\nTEST_7Z_PATH = '/kaggle/input/competitions/malware-classification/test.7z'\nTEMP_DIR = '/kaggle/working/temp_extract_test'\nOUTPUT_DIR = '/kaggle/working/dataset_upload_test'\n\n# ==============================================================================\n# TỪ KHÓA TRÍCH XUẤT (Regex C-backend)\n# ==============================================================================\nSECTIONS = [b'.text', b'.data', b'.bss', b'.rdata', b'.edata', b'.idata', b'.rsrc', b'.tls', b'.reloc']\nOPCODES = [b'jmp', b'mov', b'retf', b'push', b'pop', b'xor', b'retn', b'nop', b'sub', b'inc', b'dec', b'add', b'imul', b'xchg', b'or', b'shr', b'cmp', b'call', b'shl', b'ror', b'rol', b'jnb']\nAPIS = [b'GetProcAddress', b'LoadLibrary', b'VirtualAlloc', b'WriteFile', b'CreateFile', b'RegOpenKey', b'ExitProcess', b'GetModuleHandle', b'InternetOpen', b'CreateThread', b'Sleep', b'FindResource', b'CloseHandle']\nREGISTERS = [b'eax', b'ebx', b'ecx', b'edx', b'esi', b'edi', b'esp', b'ebp']\n\nRE_SECTION = re.compile(br'^\\.([a-z]+)')\nRE_KEYWORDS = re.compile(b'(?i)\\\\b(' + b'|'.join(OPCODES + APIS + REGISTERS) + b')\\\\b')\n\n# ==============================================================================\n# HÀM XỬ LÝ LÕI\n# ==============================================================================\ndef process_single_test_malware(file_id):\n    bytes_path = os.path.join(TEMP_DIR, f\"{file_id}.bytes\")\n    asm_path = os.path.join(TEMP_DIR, f\"{file_id}.asm\")\n    \n    feat = {'Id': file_id}\n    \n    # 1. Đo kích thước\n    feat['size_bytes_MB'] = os.path.getsize(bytes_path) / (1024*1024) if os.path.exists(bytes_path) else 0\n    feat['size_asm_MB'] = os.path.getsize(asm_path) / (1024*1024) if os.path.exists(asm_path) else 0\n    \n    sec_counts = {s.decode(): 0 for s in SECTIONS}\n    keyword_counts = {k.decode().lower(): 0 for k in OPCODES + APIS + REGISTERS}\n    \n    # 2. Quét tốc độ cao\n    if os.path.exists(asm_path):\n        try:\n            with open(asm_path, 'rb') as f:\n                for line in f:\n                    sec_match = RE_SECTION.search(line)\n                    if sec_match:\n                        sec_name = b'.' + sec_match.group(1)\n                        if sec_name in SECTIONS:\n                            sec_counts[sec_name.decode()] += 1\n                    \n                    kw_matches = RE_KEYWORDS.findall(line)\n                    for kw in kw_matches:\n                        keyword_counts[kw.decode().lower()] += 1\n        except Exception:\n            pass \n\n    # 3. Tổng hợp Features\n    for k, v in sec_counts.items():\n        feat[f\"sec_{k.replace('.','')}\"] = v\n    for k, v in keyword_counts.items():\n        feat[f\"kw_{k}\"] = v\n        \n    text_c = feat.get('sec_text', 0) + 1\n    feat['ratio_data_text'] = feat.get('sec_data', 0) / text_c\n    feat['ratio_rdata_text'] = feat.get('sec_rdata', 0) / text_c\n\n    return feat\n\ndef extract_test_batch(file_ids):\n    os.makedirs(TEMP_DIR, exist_ok=True)\n    list_file = '/kaggle/working/batch_include_test.txt'\n    with open(list_file, 'w') as f:\n        for fid in file_ids: \n            # Kaggle test.7z chứa thư mục 'test/'\n            f.write(f\"test/{fid}.bytes\\n\")\n            f.write(f\"test/{fid}.asm\\n\")\n            \n    # Lệnh giải nén hàng loạt\n    cmd = ['7z', 'e', TEST_7Z_PATH, f'-o{TEMP_DIR}', f'-i@{list_file}', '-y']\n    subprocess.run(cmd, stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL)\n    if os.path.exists(list_file): os.remove(list_file)\n\ndef upload_test_dataset(df_result):\n    os.makedirs(OUTPUT_DIR, exist_ok=True)\n    csv_path = os.path.join(OUTPUT_DIR, f'test_metadata_part_{MY_PART}.csv')\n    df_result.to_csv(csv_path, index=False)\n    \n    dataset_slug = f'malware-tabular-test-p{MY_PART}'\n    dataset_id = f\"{KAGGLE_USER}/{dataset_slug}\"\n\n    metadata = {\n      \"title\": f'Malware Tabular Test P{MY_PART}',\n      \"id\": dataset_id,\n      \"licenses\": [{\"name\": \"CC0-1.0\"}]\n    }\n    with open(f'{OUTPUT_DIR}/dataset-metadata.json', 'w') as f:\n        json.dump(metadata, f)\n        \n    print(f\"\\n[TIẾN TRÌNH] Bắt đầu Upload Kaggle Dataset Test (Part {MY_PART})...\")\n    try:\n        api = KaggleApi()\n        api.authenticate()\n        api.dataset_create_new(folder=OUTPUT_DIR, dir_mode='zip', public=False)\n        print(\"[HOÀN THÀNH] Dataset Test CSV đã được đẩy lên thành công!\")\n    except Exception as e:\n        print(\"[LỖI UPLOAD] Không thể upload:\", e)\n        print(f\"File CSV vẫn an toàn tại: {csv_path}. Bạn hãy bấm tải thủ công (Download).\")\n\ndef main():\n    try:\n        user_secrets = UserSecretsClient()\n        os.environ['KAGGLE_USERNAME'] = user_secrets.get_secret(\"KAGGLE_USERNAME\")\n        os.environ['KAGGLE_KEY'] = user_secrets.get_secret(\"KAGGLE_KEY\")\n    except Exception:\n        print(\">> [CẢNH BÁO] Không tìm thấy API Secret. Bỏ qua tự động Upload.\")\n\n    print(\">> Đọc danh sách ID từ sampleSubmission.csv...\")\n    df_test = pd.read_csv(TEST_SUBMISSION_CSV)\n    all_test_ids = df_test['Id'].values\n    \n    # Chia làm 4 mảng bằng nhau\n    my_tasks = np.array_split(all_test_ids, TOTAL_PARTS)[MY_PART]\n    total_files = len(my_tasks)\n    \n    print(f\"\\n🚀 BẮT ĐẦU CHẠY LUỒNG 2.3 (TẬP TEST) - PART {MY_PART}/{TOTAL_PARTS} 🚀\")\n    print(f\"Tổng số ID Test cần xử lý: {total_files} | Batch Size Tối ưu: {BATCH_SIZE}\")\n    \n    pool = mp.Pool(processes=mp.cpu_count())\n    all_extracted_features = []\n    processed_count = 0\n    \n    for i in range(0, total_files, BATCH_SIZE):\n        elapsed_time = time.time() - START_TIME\n        if elapsed_time > TIMEOUT_LIMIT:\n            print(\"⏳ Đã chạm giới hạn thời gian. Dừng sớm!\")\n            break\n            \n        batch_ids = my_tasks[i : i + BATCH_SIZE]\n        \n        # 1. Giải nén 1 cục lớn\n        extract_test_batch(batch_ids)\n        \n        # 2. Xử lý đa luồng\n        for result in pool.imap_unordered(process_single_test_malware, batch_ids):\n            if result: \n                all_extracted_features.append(result)\n                processed_count += 1\n                \n        # 3. Dọn rác triệt để ngay lập tức\n        for fid in batch_ids:\n            try:\n                os.remove(os.path.join(TEMP_DIR, f\"{fid}.bytes\"))\n                os.remove(os.path.join(TEMP_DIR, f\"{fid}.asm\"))\n            except: pass\n            \n        gc.collect()\n\n        if processed_count % BATCH_SIZE == 0 or processed_count == total_files:\n            percent = (processed_count/total_files)*100\n            print(f\"[{time.strftime('%H:%M:%S', time.gmtime(elapsed_time))}] Đã xử lý: {processed_count}/{total_files} files ({percent:.1f}%)\")\n\n    pool.close()\n    pool.join()\n    \n    # Xuất file và dọn dẹp\n    df_result = pd.DataFrame(all_extracted_features)\n    print(\"\\n[MẪU DỮ LIỆU THU ĐƯỢC]\")\n    print(df_result.head(3))\n    \n    upload_test_dataset(df_result)\n\nif __name__ == '__main__':\n    main()","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"outputs":[],"execution_count":null}]}