{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":4117,"databundleVersionId":46665,"sourceType":"competition"},{"sourceId":14322821,"sourceType":"datasetVersion","datasetId":9139823,"isSourceIdPinned":true}],"dockerImageVersionId":31234,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install py7zr","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-27T19:02:20.662962Z","iopub.execute_input":"2025-12-27T19:02:20.663335Z","iopub.status.idle":"2025-12-27T19:02:45.021227Z","shell.execute_reply.started":"2025-12-27T19:02:20.663306Z","shell.execute_reply":"2025-12-27T19:02:45.019987Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport cv2\nimport pandas as pd\nimport os\nimport shutil\nimport py7zr\nimport gc\n\n# --- CẤU HÌNH ---\nSOURCE_7Z = '/kaggle/input/malware-classification/train.7z'\nTEMP_DIR = '/kaggle/working/temp_extract'\nIMG_OUTPUT = '/kaggle/working/processed_images'\n# Đảm bảo đường dẫn này trỏ đúng file train_ids.txt của bạn\nXGB_IDS_PATH = '/kaggle/input/malware-features-ckpt-v5/features_full_v4/train_ids.txt' \n\nif not os.path.exists(IMG_OUTPUT): os.makedirs(IMG_OUTPUT)\nif not os.path.exists(TEMP_DIR): os.makedirs(TEMP_DIR)\n\n# --- 1. ĐỊNH NGHĨA HÀM XỬ LÝ ẢNH (QUAN TRỌNG) ---\ndef bytes_to_image(path, save_dir):\n    try:\n        with open(path, 'r') as f:\n            arr = []\n            for line in f:\n                parts = line.split()\n                if len(parts) == 17:\n                    row_bytes = []\n                    for i in parts[1:]:\n                        if i == '??': row_bytes.append(0)\n                        else: row_bytes.append(int(i, 16))\n                    arr.append(row_bytes)\n        \n        if len(arr) == 0: return\n\n        img = np.array(arr)\n        # Logic resize về 226x226\n        target_size = (226, 226)\n        side = int(np.sqrt(img.size))\n        if side < 1: side = 1\n        \n        # Reshape thành hình vuông (gần đúng) rồi resize chuẩn\n        img_2d = img.flatten()[:side*side].reshape(side, side)\n        img_final = cv2.resize(np.uint8(img_2d), target_size)\n\n        fname = os.path.basename(path).replace('.bytes', '.png')\n        cv2.imwrite(os.path.join(save_dir, fname), img_final)\n    except Exception as e:\n        print(f\"Lỗi khi xử lý file {path}: {e}\")\n\n# --- 2. LẤY DANH SÁCH ID TỪ XGBOOST ---\nprint(\"🚀 Bắt đầu lấy danh sách file...\")\n\ntry:\n    with open(XGB_IDS_PATH, 'r') as f:\n        # Tạo set các ID cần thiết\n        target_ids = set([line.strip() for line in f if line.strip()])\n    print(f\"✅ Đã load {len(target_ids)} ID từ XGBoost dataset.\")\nexcept FileNotFoundError:\n    print(\"❌ LỖI: Không tìm thấy file train_ids.txt! Hãy kiểm tra đường dẫn XGB_IDS_PATH.\")\n    raise\n\n# --- 3. LẤY DANH SÁCH FILE TỪ 7Z ---\nwith py7zr.SevenZipFile(SOURCE_7Z, mode='r') as z:\n    all_files = z.getnames()\n\n# --- 4. LỌC FILE (CHỈ LẤY FILE CÓ TRONG XGBOOST) ---\nfile_list = []\nfor f in all_files:\n    if f.startswith('train/') and f.endswith('.bytes'):\n        # Lấy ID từ tên file\n        file_id = os.path.basename(f).replace('.bytes', '')\n        \n        # Chỉ thêm vào danh sách xử lý nếu ID này nằm trong tập XGBoost\n        if file_id in target_ids:\n            file_list.append(f)\n\nprint(f\"--> Tìm thấy {len(file_list)} file '.bytes' khớp với XGBoost. Bắt đầu xử lý...\")\n\n# --- 5. CHẠY BATCH PROCESSING ---\nBATCH_SIZE = 100 \nprocessed_count = 0\n\nfor i in range(0, len(file_list), BATCH_SIZE):\n    batch_files = file_list[i : i + BATCH_SIZE]\n    \n    print(f\"⏳ Batch {i//BATCH_SIZE + 1}: Giải nén {len(batch_files)} file...\")\n    \n    try:\n        with py7zr.SevenZipFile(SOURCE_7Z, mode='r') as z_batch:\n            z_batch.extract(targets=batch_files, path=TEMP_DIR)\n        \n        extract_path = os.path.join(TEMP_DIR, 'train')\n        if not os.path.exists(extract_path): extract_path = TEMP_DIR\n        \n        current_files = os.listdir(extract_path)\n        for f in current_files:\n            f_path = os.path.join(extract_path, f)\n            if f.endswith('.bytes'):\n                # GỌI HÀM ĐÃ ĐỊNH NGHĨA Ở TRÊN\n                bytes_to_image(f_path, IMG_OUTPUT)\n        \n        # Dọn dẹp\n        shutil.rmtree(TEMP_DIR)\n        os.makedirs(TEMP_DIR)\n        processed_count += len(batch_files)\n        gc.collect()\n        \n    except Exception as e:\n        print(f\"⚠️ Lỗi ở Batch {i//BATCH_SIZE + 1}: {e}\")\n        if os.path.exists(TEMP_DIR):\n            shutil.rmtree(TEMP_DIR)\n            os.makedirs(TEMP_DIR)\n        continue\n\nprint(f\"\\n✅ Hoàn tất! Đã xử lý {processed_count} ảnh vào {IMG_OUTPUT}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-28T04:14:38.448751Z","iopub.execute_input":"2025-12-28T04:14:38.449093Z","iopub.status.idle":"2025-12-28T04:14:40.333333Z","shell.execute_reply.started":"2025-12-28T04:14:38.449052Z","shell.execute_reply":"2025-12-28T04:14:40.331462Z"}},"outputs":[],"execution_count":null}]}