{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceType":"competition","sourceId":4117,"databundleVersionId":46665}],"dockerImageVersionId":31328,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install py7zr","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2026-04-07T15:51:42.084901Z","iopub.execute_input":"2026-04-07T15:51:42.085313Z","iopub.status.idle":"2026-04-07T15:51:45.436109Z","shell.execute_reply.started":"2026-04-07T15:51:42.08527Z","shell.execute_reply":"2026-04-07T15:51:45.435186Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import py7zr\nimport numpy as np\nimport cv2\nimport matplotlib.pyplot as plt\nfrom tqdm import tqdm\nimport os","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-07T15:51:45.438891Z","iopub.execute_input":"2026-04-07T15:51:45.43921Z","iopub.status.idle":"2026-04-07T15:51:45.551482Z","shell.execute_reply.started":"2026-04-07T15:51:45.439179Z","shell.execute_reply":"2026-04-07T15:51:45.550399Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def parse_bytes_file(file_content):\n    \"\"\"Đọc nội dung thô của file .bytes và trả về mảng 1D numpy chứa các số nguyên 0-255.\"\"\"\n    hex_data = []\n    # file_content đọc từ zip ra sẽ ở dạng bytes, cần decode\n    lines = file_content.decode('utf-8').splitlines()\n    \n    for line in lines:\n        tokens = line.split()\n        if len(tokens) > 1:\n            # Bỏ qua tokens[0] (địa chỉ), chỉ lấy giá trị byte\n            for token in tokens[1:]:\n                if token == '??':\n                    hex_data.append(0)\n                else:\n                    hex_data.append(int(token, 16))\n                    \n    return np.array(hex_data, dtype=np.uint8)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-07T15:51:45.553086Z","iopub.execute_input":"2026-04-07T15:51:45.553537Z","iopub.status.idle":"2026-04-07T15:51:45.561937Z","shell.execute_reply.started":"2026-04-07T15:51:45.553508Z","shell.execute_reply":"2026-04-07T15:51:45.560495Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def to_grayscale_image(hex_array, width=256):\n    # Tính toán chiều cao sao cho vừa với width\n    height = len(hex_array) // width\n    \n    # Cắt bỏ phần dư ở cuối (nếu có) để reshape thành mảng 2D hoàn chỉnh\n    img_array = hex_array[:height * width].reshape((height, width))\n    return img_array\n\ndef to_rgb_image(hex_array, width=256):\n    # Một hàng có 'width' pixels, mỗi pixel cần 3 bytes -> bytes_per_row = width * 3\n    bytes_per_row = width * 3\n    height = len(hex_array) // bytes_per_row\n    \n    # Cắt phần dư và reshape thành mảng 3D (H, W, 3)\n    img_array = hex_array[:height * bytes_per_row].reshape((height, width, 3))\n    return img_array\n\ndef to_bigram_matrix_image(hex_array):\n    # Khởi tạo ma trận 256x256 với giá trị 0\n    matrix = np.zeros((256, 256), dtype=np.float64)\n    \n    # Kỹ thuật đếm nhanh bằng numpy:\n    # hex_array[:-1] là mảng các byte thứ i\n    # hex_array[1:] là mảng các byte thứ i+1 (ngay sau i)\n    np.add.at(matrix, (hex_array[:-1], hex_array[1:]), 1)\n    \n    # Lấy logarit để các tần suất nhỏ không bị lu mờ bởi các tần suất quá lớn\n    matrix = np.log1p(matrix)\n    \n    # Chuẩn hóa về thang 0 - 255 để tạo thành ảnh\n    if matrix.max() > 0:\n        matrix = (matrix / matrix.max()) * 255.0\n        \n    return matrix.astype(np.uint8)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-07T15:51:45.562679Z","iopub.execute_input":"2026-04-07T15:51:45.562952Z","iopub.status.idle":"2026-04-07T15:51:45.581701Z","shell.execute_reply.started":"2026-04-07T15:51:45.562924Z","shell.execute_reply":"2026-04-07T15:51:45.581045Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport subprocess\nimport cv2\nfrom tqdm import tqdm\nimport py7zr\nimport time\nimport shutil\nimport json\nfrom kaggle_secrets import UserSecretsClient\nfrom kaggle.api.kaggle_api_extended import KaggleApi\n\n# --- 1. CẤU HÌNH ĐƯỜNG DẪN & THỜI GIAN ---\nstart_time = time.time()\nMAX_RUNTIME_SECONDS = 11.5 * 3600 # Ngắt sau 11.5 tiếng\n\narchive_path = '/kaggle/input/competitions/malware-classification/test.7z'\nbase_output_dir = '/kaggle/working/processed_images/'\ntemp_dir = '/kaggle/working/temp_bytes/'\n\n# ĐƯỜNG DẪN DATASET LẦN 1 (Sửa 'malware-processed-images-v1' thành tên đúng của bạn ở Bước 1)\nold_dataset_path = '/kaggle/input/malware-processed-images-v1/' \n\nsub_dirs = {\n    'gray': os.path.join(base_output_dir, 'gray'),\n    'rgb': os.path.join(base_output_dir, 'rgb'),\n    'bigram': os.path.join(base_output_dir, 'bigram')\n}\n\n# Tạo thư mục working\nos.makedirs(temp_dir, exist_ok=True)\nfor path in sub_dirs.values():\n    os.makedirs(path, exist_ok=True)\n\n# --- 2. COPY DỮ LIỆU CŨ TỪ LẦN 1 SANG WORKING ---\n# Mục đích: Gom cả cũ và mới vào chung một chỗ để upload thành Version 2\nif os.path.exists(old_dataset_path) and len(os.listdir(old_dataset_path)) > 0:\n    print(f\"Đang copy dữ liệu cũ từ {old_dataset_path} sang working...\")\n    # Lệnh copy linux chạy rất nhanh\n    os.system(f'cp -rn {old_dataset_path}/* {base_output_dir}/')\nelse:\n    print(\"Không tìm thấy Dataset cũ hoặc Dataset trống, sẽ chạy từ đầu.\")\n\n# --- 3. KIỂM TRA NHỮNG FILE ĐÃ CHẠY ---\nprocessed_files = set()\nif os.path.exists(sub_dirs['gray']):\n    # Lấy danh sách các file .png đã có trong folder gray và đổi tên lại thành .bytes để đối chiếu\n    processed_files = {f.replace('.png', '.bytes') for f in os.listdir(sub_dirs['gray'])}\n\n# Đọc danh sách file gốc\nwith py7zr.SevenZipFile(archive_path, mode='r') as z:\n    all_files = z.getnames()\n    bytes_files = [f for f in all_files if f.endswith('.bytes')]\n\n# Lọc ra những file CHƯA chạy\nfiles_to_process = [f for f in bytes_files if f not in processed_files]\nprint(f\"Tổng số file gốc: {len(bytes_files)}\")\nprint(f\"Đã xử lý (từ Lần trước): {len(processed_files)}\")\nprint(f\"Cần xử lý tiếp trong phiên này: {len(files_to_process)}\")\n\n# --- 4. VÒNG LẶP XỬ LÝ (Có điều kiện ngắt thời gian) ---\nfor fname in tqdm(files_to_process):\n    # Kiểm tra thời gian\n    if time.time() - start_time > MAX_RUNTIME_SECONDS:\n        print(f\"\\n[CẢNH BÁO] Đã chạm ngưỡng {MAX_RUNTIME_SECONDS/3600} tiếng! Đang dừng an toàn...\")\n        break # Thoát vòng lặp, đi xuống code Upload\n        \n    # Code xử lý giải nén, chuyển ảnh của bạn (GIỮ NGUYÊN)\n    cmd = ['7z', 'e', archive_path, f'-o{temp_dir}', fname, '-y']\n    subprocess.run(cmd, stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL)\n    \n    base_name = os.path.basename(fname)\n    extracted_path = os.path.join(temp_dir, base_name)\n    \n    if os.path.exists(extracted_path):\n        with open(extracted_path, 'rb') as f:\n            raw_content = f.read()\n            \n        hex_array = parse_bytes_file(raw_content) # Hàm của bạn\n        \n        img_gray = to_grayscale_image(hex_array) # Hàm của bạn\n        img_rgb = to_rgb_image(hex_array) # Hàm của bạn\n        img_bigram = to_bigram_matrix_image(hex_array) # Hàm của bạn\n        \n        file_id = base_name.replace('.bytes', '')\n        filename = f\"{file_id}.png\"\n        \n        cv2.imwrite(os.path.join(sub_dirs['gray'], filename), img_gray)\n        cv2.imwrite(os.path.join(sub_dirs['rgb'], filename), cv2.cvtColor(img_rgb, cv2.COLOR_RGB2BGR))\n        cv2.imwrite(os.path.join(sub_dirs['bigram'], filename), img_bigram)\n        \n        os.remove(extracted_path)\n\nprint(\"Kết thúc quá trình xử lý. Bắt đầu Upload Version mới lên Dataset...\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-07T15:51:45.586129Z","iopub.execute_input":"2026-04-07T15:51:45.588899Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport json\nfrom kaggle_secrets import UserSecretsClient\n\n# 1. Lấy thông tin xác thực từ Kaggle Secrets\ntry:\n    user_secrets = UserSecretsClient()\n    os.environ['KAGGLE_USERNAME'] = user_secrets.get_secret(\"KAGGLE_USERNAME\")\n    os.environ['KAGGLE_KEY'] = user_secrets.get_secret(\"KAGGLE_KEY\")\nexcept Exception as e:\n    print(\"Vui lòng thiết lập Kaggle Secrets trước:\", e)\n\nfrom kaggle.api.kaggle_api_extended import KaggleApi\n\n# 2. Khởi tạo API\napi = KaggleApi()\napi.authenticate()\n\n# 3. Cấu hình Dataset\ndataset_dir = '/kaggle/working/processed_images/'\ndataset_title = 'Malware Processed Images - Test' # Tên hiển thị của Dataset\ndataset_slug = 'malware-processed-images-v1' # Viết thường, không dấu, cách nhau bằng gạch ngang\n\n# LƯU Ý: Thay 'YOUR_KAGGLE_USERNAME' bằng username thật của bạn\nusername = os.environ['KAGGLE_USERNAME']\ndataset_id = f\"{username}/{dataset_slug}\"\n\n# 4. Tạo file metadata bắt buộc cho Kaggle Dataset\nmetadata = {\n  \"title\": dataset_title,\n  \"id\": dataset_id,\n  \"licenses\": [{\"name\": \"CC0-1.0\"}]\n}\n\nwith open(os.path.join(dataset_dir, 'dataset-metadata.json'), 'w') as f:\n    json.dump(metadata, f)\n\n# 5. Tự động Upload tạo Dataset\nprint(f\"Bắt đầu upload dataset: {dataset_id}...\")\n\ntry:\n    # Nếu tạo lần đầu:\n    api.dataset_create_new(folder=dataset_dir, dir_mode='zip', public=False)\n    print(\"Đã tạo Dataset mới thành công! Bạn có thể vào phần Datasets trên Kaggle để kiểm tra.\")\n    \n    # Nếu Dataset đã tồn tại và bạn muốn cập nhật (bỏ comment dòng dưới, comment dòng trên):\n    # api.dataset_create_version(folder=dataset_dir, version_notes=\"Cập nhật thêm data\", dir_mode='zip')\n    # print(\"Đã cập nhật Dataset thành công!\")\n    \nexcept Exception as e:\n    print(\"Có lỗi xảy ra khi upload Dataset:\", e)","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}