{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceType":"competition","sourceId":4117,"databundleVersionId":46665}],"dockerImageVersionId":31328,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"pip install py7zr","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2026-04-24T07:14:38.265173Z","iopub.execute_input":"2026-04-24T07:14:38.265531Z","iopub.status.idle":"2026-04-24T07:14:41.509856Z","shell.execute_reply.started":"2026-04-24T07:14:38.265496Z","shell.execute_reply":"2026-04-24T07:14:41.508688Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\n# Load data's labels\nlabels_csv = \"/kaggle/input/competitions/malware-classification/trainLabels.csv\"\nlabels = pd.read_csv(labels_csv)\n\n# Count number of classes\nclass_count = labels['Class'].value_counts().sort_index()\nprint(class_count)\n\n# Visualize data\nplt.figure(figsize = (10 , 5))\nsns.countplot(x = 'Class' , data = labels)\nplt.title(\"Số lượng mã độc cho từng Class\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-24T07:14:41.511705Z","iopub.execute_input":"2026-04-24T07:14:41.511922Z","iopub.status.idle":"2026-04-24T07:14:41.87293Z","shell.execute_reply.started":"2026-04-24T07:14:41.511899Z","shell.execute_reply":"2026-04-24T07:14:41.871643Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ============================================================\n# CODE SPACE 1 — Đọc trainLabels.csv\n# ============================================================\n\nimport pandas as pd\n\nLABELS_CSV = \"/kaggle/input/competitions/malware-classification/trainLabels.csv\"\n\ndf_labels = pd.read_csv(LABELS_CSV)\ndf_labels.columns = df_labels.columns.str.strip()\n\n# dict: { \"Hash123\": \"1\", \"Hash456\": \"3\", ... }\nlabel_map = dict(zip(df_labels[\"Id\"].str.strip(), df_labels[\"Class\"].astype(str).str.strip()))\n\nprint(f\"✅ Đọc được {len(label_map)} labels\")\nprint(f\"   Classes: {sorted(df_labels['Class'].unique())}\")\nprint(df_labels[\"Class\"].value_counts().sort_index().rename(\"count\").to_frame())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-24T07:14:41.874647Z","iopub.execute_input":"2026-04-24T07:14:41.874974Z","iopub.status.idle":"2026-04-24T07:14:41.927133Z","shell.execute_reply.started":"2026-04-24T07:14:41.874943Z","shell.execute_reply":"2026-04-24T07:14:41.926459Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ============================================================\n# CODE SPACE 2 — Đọc thẳng từ .7z → convert sang ảnh PNG\n# Không giải nén ra disk — tiết kiệm ~10GB dung lượng\n# Yêu cầu: đã chạy Code space 1 (có label_map)\n# ============================================================\n\nimport os\nimport math\nimport io\nimport subprocess\nfrom pathlib import Path\nfrom concurrent.futures import ProcessPoolExecutor, as_completed\n\nimport numpy as np\nfrom PIL import Image\nfrom tqdm import tqdm\n\n# ── Config ────────────────────────────────────────────────\nINPUT_DIR  = \"/kaggle/input/competitions/malware-classification\"\nTRAIN_7Z   = f\"{INPUT_DIR}/train.7z\"\nTEST_7Z    = f\"{INPUT_DIR}/test.7z\"\nTRAIN_OUT  = \"/kaggle/working/train_images\"   # class subfolders\nTEST_OUT   = \"/kaggle/working/test_images\"    # flat\nIMG_SIZE   = 256\nNUM_WORKERS = os.cpu_count() or 2\n\nos.makedirs(TRAIN_OUT, exist_ok=True)\nos.makedirs(TEST_OUT,  exist_ok=True)\n\nprint(f\"⚙️  IMG_SIZE={IMG_SIZE}  |  workers={NUM_WORKERS}\")\n\n\n# ── Core: bytes content → PNG ──────────────────────────────\n\ndef content_to_image(content: bytes, output_path: str, img_size: int = 256) -> bool:\n    \"\"\"Chuyển nội dung file .bytes (đã đọc vào memory) → PNG grayscale.\"\"\"\n    pixels = []\n    try:\n        for line in content.decode(\"utf-8\", errors=\"ignore\").splitlines():\n            tokens = line.split()\n            for tok in tokens[1:]:        # bỏ cột địa chỉ đầu dòng\n                if tok == \"??\":\n                    pixels.append(0)\n                else:\n                    try:\n                        pixels.append(int(tok, 16))\n                    except ValueError:\n                        pass\n    except Exception:\n        return False\n\n    if not pixels:\n        return False\n\n    arr   = np.array(pixels, dtype=np.uint8)\n    side  = img_size if img_size else math.ceil(math.sqrt(len(arr)))\n    total = side * side\n    arr   = np.pad(arr, (0, max(0, total - len(arr))))[:total]\n\n    os.makedirs(os.path.dirname(output_path), exist_ok=True)\n    Image.fromarray(arr.reshape(side, side), mode=\"L\").save(output_path, optimize=True)\n    return True\n\n\n# ── Đọc file list trong .7z (không giải nén) ──────────────\n\ndef list_7z(archive: str) -> list[str]:\n    \"\"\"Trả về danh sách tên file bên trong archive.\"\"\"\n    result = subprocess.run(\n        [\"7z\", \"l\", \"-ba\", \"-slt\", archive],\n        capture_output=True, text=True\n    )\n    files = []\n    for line in result.stdout.splitlines():\n        if line.startswith(\"Path = \"):\n            name = line[7:].strip()\n            if name.endswith(\".bytes\"):\n                files.append(name)\n    return files\n\n\n# ── Đọc 1 file từ .7z vào memory ──────────────────────────\n\ndef read_one_from_7z(archive: str, filename: str) -> bytes | None:\n    \"\"\"Giải nén 1 file vào RAM, không ghi ra disk.\"\"\"\n    result = subprocess.run(\n        [\"7z\", \"e\", archive, filename, \"-so\"],   # -so = stdout\n        capture_output=True\n    )\n    if result.returncode != 0 or not result.stdout:\n        return None\n    return result.stdout\n\n\n# ── Worker (chạy trong subprocess) ────────────────────────\n\ndef _worker_train(args: tuple) -> bool:\n    archive, filename, out_path, img_size = args\n    if os.path.exists(out_path):\n        return True\n    content = read_one_from_7z(archive, filename)\n    if content is None:\n        return False\n    return content_to_image(content, out_path, img_size)\n\n\ndef _worker_test(args: tuple) -> bool:\n    archive, filename, out_path, img_size = args\n    if os.path.exists(out_path):\n        return True\n    content = read_one_from_7z(archive, filename)\n    if content is None:\n        return False\n    return content_to_image(content, out_path, img_size)\n\n\n# ── Runner ─────────────────────────────────────────────────\n\ndef run_parallel(tasks: list, worker_fn, desc: str):\n    ok = fail = skip = 0\n    with ProcessPoolExecutor(max_workers=NUM_WORKERS) as pool:\n        futures = {pool.submit(worker_fn, t): t for t in tasks}\n        with tqdm(total=len(tasks), desc=desc, unit=\"img\") as bar:\n            for fut in as_completed(futures):\n                result = fut.result()\n                if result is True:\n                    ok += 1\n                else:\n                    fail += 1\n                bar.update(1)\n    print(f\"   ✅ {ok} ảnh  {'| ❌ ' + str(fail) + ' lỗi' if fail else ''}\\n\")\n\n\n# ── TRAIN ─────────────────────────────────────────────────\n\nprint(\"\\n── TRAIN ──\")\nprint(\"   Đọc danh sách file trong train.7z ...\")\ntrain_files = list_7z(TRAIN_7Z)\nprint(f\"   Tìm thấy {len(train_files)} file .bytes\")\n\ntrain_tasks = []\nmissing_label = []\nfor fname in train_files:\n    stem = Path(fname).stem\n    cls  = label_map.get(stem)        # label_map từ Code space 1\n    if cls is None:\n        missing_label.append(stem)\n        continue\n    out  = os.path.join(TRAIN_OUT, cls, f\"{stem}.png\")\n    train_tasks.append((TRAIN_7Z, fname, out, IMG_SIZE))\n\nif missing_label:\n    print(f\"   ⚠️  {len(missing_label)} file không có label → bỏ qua\")\n\nprint(f\"   Cần convert: {len(train_tasks)} files\")\nrun_parallel(train_tasks, _worker_train, \"🖼️  Train\")\n\n\n# ── TEST ──────────────────────────────────────────────────\n\nprint(\"── TEST ──\")\nprint(\"   Đọc danh sách file trong test.7z ...\")\ntest_files = list_7z(TEST_7Z)\nprint(f\"   Tìm thấy {len(test_files)} file .bytes\")\n\ntest_tasks = [\n    (TEST_7Z, fname, os.path.join(TEST_OUT, f\"{Path(fname).stem}.png\"), IMG_SIZE)\n    for fname in test_files\n]\n\nprint(f\"   Cần convert: {len(test_tasks)} files\")\nrun_parallel(test_tasks, _worker_test, \"🧪 Test \")\n\n\n# ── Tổng kết ──────────────────────────────────────────────\n\nprint(\"=\" * 45)\nfor label, folder in [(\"Train\", TRAIN_OUT), (\"Test\", TEST_OUT)]:\n    pngs = list(Path(folder).rglob(\"*.png\"))\n    mb   = sum(f.stat().st_size for f in pngs) / 1e6\n    print(f\"  {label:5}: {len(pngs):>6} ảnh  ({mb:>7.1f} MB)  →  {folder}\")\n\nprint(\"\\n✅ Code space 2 hoàn tất!\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-24T07:14:41.92877Z","iopub.execute_input":"2026-04-24T07:14:41.92897Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ============================================================\n# CODE SPACE 3 — Kiểm tra & visualize kết quả\n# ============================================================\n\nimport random\nfrom pathlib import Path\nfrom collections import Counter\n\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport matplotlib.gridspec as gridspec\nfrom PIL import Image, UnidentifiedImageError\n\n# ── Config (fallback nếu chạy cell độc lập) ───────────────\ntry:\n    TRAIN_OUT, TEST_OUT\nexcept NameError:\n    TRAIN_OUT = \"/kaggle/working/train_images\"\n    TEST_OUT  = \"/kaggle/working/test_images\"\n\n\n# ── 1. Thống kê số ảnh mỗi class ──────────────────────────\n\nprint(\"📊 Train — số ảnh mỗi class:\")\ncounts = Counter()\nfor d in sorted(Path(TRAIN_OUT).iterdir()):\n    if d.is_dir():\n        counts[d.name] = len(list(d.glob(\"*.png\")))\n\ntotal = sum(counts.values())\nfor cls, n in sorted(counts.items()):\n    bar = \"█\" * (n // 150)\n    print(f\"  Class {cls:>2}: {n:>5} ảnh  ({n/total*100:4.1f}%)  {bar}\")\nprint(f\"  {'Tổng':>7}: {total:>5} ảnh\\n\")\n\ntest_pngs = list(Path(TEST_OUT).glob(\"*.png\"))\nprint(f\"📊 Test: {len(test_pngs)} ảnh\\n\")\n\n\n# ── 2. Kiểm tra ảnh corrupt ────────────────────────────────\n\nprint(\"🔍 Kiểm tra ảnh corrupt (sample 300 files)...\")\nall_pngs = list(Path(TRAIN_OUT).rglob(\"*.png\"))\nsample   = random.sample(all_pngs, min(300, len(all_pngs)))\nbad = []\nfor f in sample:\n    try:\n        Image.open(f).verify()\n    except Exception:\n        bad.append(str(f))\n\nif bad:\n    print(f\"  ⚠️  {len(bad)} ảnh lỗi:\")\n    for b in bad[:5]: print(f\"    {b}\")\nelse:\n    print(\"  ✅ Không có ảnh lỗi\\n\")\n\n\n# ── 3. Visualize mẫu 3 ảnh mỗi class ─────────────────────\n\nprint(\"🖼️  Visualize mẫu ảnh mỗi class...\")\nclass_dirs = sorted([d for d in Path(TRAIN_OUT).iterdir() if d.is_dir()])\nN, C = len(class_dirs), 3\n\nfig = plt.figure(figsize=(C * 3.2, N * 3.2))\nfig.suptitle(\"Malware Visualization — Grayscale (.bytes → PNG)\",\n             fontsize=13, fontweight=\"bold\", y=1.01)\ngs = gridspec.GridSpec(N, C, hspace=0.4, wspace=0.1)\n\nfor r, cls_dir in enumerate(class_dirs):\n    imgs    = list(cls_dir.glob(\"*.png\"))\n    samples = random.sample(imgs, min(C, len(imgs)))\n    for c in range(C):\n        ax = fig.add_subplot(gs[r, c])\n        if c < len(samples):\n            ax.imshow(np.array(Image.open(samples[c])), cmap=\"gray\", vmin=0, vmax=255)\n            if c == 0:\n                ax.set_ylabel(f\"Class {cls_dir.name}\", fontsize=10, fontweight=\"bold\")\n            ax.set_title(samples[c].stem[:14] + \"…\", fontsize=7, color=\"gray\")\n        ax.axis(\"off\")\n\nplt.savefig(\"/kaggle/working/train_samples.png\", dpi=120, bbox_inches=\"tight\")\nplt.show()\nprint(\"   Đã lưu: /kaggle/working/train_samples.png\\n\")\n\n\n# ── 4. Pixel intensity distribution mỗi class ─────────────\n\nprint(\"📈 Phân bố pixel intensity mỗi class...\")\nfig, axes = plt.subplots(3, 3, figsize=(12, 9))\nfig.suptitle(\"Pixel Intensity Distribution per Class\", fontsize=12, fontweight=\"bold\")\n\nfor ax, cls_dir in zip(axes.flat, class_dirs):\n    imgs   = list(cls_dir.glob(\"*.png\"))\n    sample = random.sample(imgs, min(200, len(imgs)))\n    pixels = np.concatenate([np.array(Image.open(p)).flatten() for p in sample])\n    ax.hist(pixels, bins=64, color=\"#2196F3\", alpha=0.8, edgecolor=\"none\")\n    ax.set_title(f\"Class {cls_dir.name}  (n={len(imgs)})\", fontsize=9)\n    ax.set_xlim(0, 255)\n    ax.set_yticks([])\n    ax.spines[[\"top\",\"right\",\"left\"]].set_visible(False)\n\nfor ax in axes.flat[len(class_dirs):]:\n    ax.set_visible(False)\n\nplt.tight_layout()\nplt.savefig(\"/kaggle/working/intensity_dist.png\", dpi=120, bbox_inches=\"tight\")\nplt.show()\nprint(\"   Đã lưu: /kaggle/working/intensity_dist.png\\n\")\n\nprint(\"✅ Dataset sẵn sàng để train model!\")\n\n","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}