{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[],"dockerImageVersionId":28755,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"**ESM**","metadata":{}},{"cell_type":"code","source":"import os\nimport time\nimport random\nimport joblib\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\n\nDATA_ROOT = \"/kaggle/input/datasets/kaso45/ember2024/vectorized_thrember_dataset\"\n\nprint(\"Dataset tồn tại:\", os.path.exists(DATA_ROOT))\nprint(\"Các file trong dataset:\")\nprint(os.listdir(DATA_ROOT))","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2026-06-18T06:10:35.933554Z","iopub.execute_input":"2026-06-18T06:10:35.933750Z","iopub.status.idle":"2026-06-18T06:10:37.295289Z","shell.execute_reply.started":"2026-06-18T06:10:35.933725Z","shell.execute_reply":"2026-06-18T06:10:37.294394Z"},"jupyter":{"source_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Đọc dữ liệu .dat**","metadata":{}},{"cell_type":"code","source":"def load_dat_dataset(data_root, subset):\n    \"\"\"\n    Đọc dữ liệu EMBER2024 dạng .dat.\n    X: vector đặc trưng, kiểu float32\n    y: nhãn, kiểu int32\n    \"\"\"\n    X_path = os.path.join(data_root, f\"X_{subset}.dat\")\n    y_path = os.path.join(data_root, f\"y_{subset}.dat\")\n    \n    if not os.path.exists(X_path):\n        raise FileNotFoundError(f\"Không tìm thấy file: {X_path}\")\n        \n    if not os.path.exists(y_path):\n        raise FileNotFoundError(f\"Không tìm thấy file: {y_path}\")\n    \n    y = np.memmap(y_path, dtype=np.int32, mode=\"r\")\n    n_rows = len(y)\n    \n    total_float32 = os.path.getsize(X_path) // np.dtype(np.float32).itemsize\n    n_features = total_float32 // n_rows\n    \n    X = np.memmap(\n        X_path,\n        dtype=np.float32,\n        mode=\"r\",\n        shape=(n_rows, n_features)\n    )\n    \n    return X, y\n\n\nX_train, y_train = load_dat_dataset(DATA_ROOT, \"train\")\nX_test, y_test = load_dat_dataset(DATA_ROOT, \"test\")\nX_challenge, y_challenge = load_dat_dataset(DATA_ROOT, \"challenge\")\n\nprint(\"X_train:\", X_train.shape)\nprint(\"y_train:\", y_train.shape)\n\nprint(\"X_test:\", X_test.shape)\nprint(\"y_test:\", y_test.shape)\n\nprint(\"X_challenge:\", X_challenge.shape)\nprint(\"y_challenge:\", y_challenge.shape)\n\nprint(\"\\nPhân bố nhãn train:\", np.unique(y_train, return_counts=True))\nprint(\"Phân bố nhãn test:\", np.unique(y_test, return_counts=True))\nprint(\"Phân bố nhãn challenge:\", np.unique(y_challenge, return_counts=True))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-18T06:10:37.297044Z","iopub.execute_input":"2026-06-18T06:10:37.297494Z","iopub.status.idle":"2026-06-18T06:10:37.379578Z","shell.execute_reply.started":"2026-06-18T06:10:37.297469Z","shell.execute_reply":"2026-06-18T06:10:37.378738Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Lấy mẫu nhỏ để chạy nhanh","metadata":{}},{"cell_type":"code","source":"N_TRAIN = 500000\nN_TEST = 100000\n\nrng = np.random.default_rng(42)\n\n# Chỉ lấy nhãn hợp lệ: 0 = Benign, 1 = Malware\nvalid_train_idx = np.where(np.isin(y_train, [0, 1]))[0]\nvalid_test_idx = np.where(np.isin(y_test, [0, 1]))[0]\n\ntrain_idx = rng.choice(\n    valid_train_idx,\n    size=min(N_TRAIN, len(valid_train_idx)),\n    replace=False\n)\n\ntest_idx = rng.choice(\n    valid_test_idx,\n    size=min(N_TEST, len(valid_test_idx)),\n    replace=False\n)\n\nX_train_small = np.array(X_train[train_idx])\ny_train_small = np.array(y_train[train_idx])\n\nX_test_small = np.array(X_test[test_idx])\ny_test_small = np.array(y_test[test_idx])\n\nprint(\"Train demo:\", X_train_small.shape)\nprint(\"Test demo:\", X_test_small.shape)\n\nprint(\"Phân bố nhãn train demo:\", np.unique(y_train_small, return_counts=True))\nprint(\"Phân bố nhãn test demo:\", np.unique(y_test_small, return_counts=True))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-18T06:10:37.380573Z","iopub.execute_input":"2026-06-18T06:10:37.380742Z","iopub.status.idle":"2026-06-18T06:12:02.330626Z","shell.execute_reply.started":"2026-06-18T06:10:37.380726Z","shell.execute_reply":"2026-06-18T06:12:02.329896Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install lightgbm -q","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-18T06:12:02.331582Z","iopub.execute_input":"2026-06-18T06:12:02.331828Z","iopub.status.idle":"2026-06-18T06:12:06.692738Z","shell.execute_reply.started":"2026-06-18T06:12:02.331808Z","shell.execute_reply":"2026-06-18T06:12:06.691676Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.pipeline import Pipeline\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.tree import DecisionTreeClassifier\nfrom sklearn.ensemble import RandomForestClassifier, ExtraTreesClassifier\n\nfrom lightgbm import LGBMClassifier\n\nfrom sklearn.metrics import (\n    accuracy_score,\n    precision_score,\n    recall_score,\n    f1_score,\n    classification_report,\n    confusion_matrix,\n    ConfusionMatrixDisplay\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-18T06:12:06.695482Z","iopub.execute_input":"2026-06-18T06:12:06.695767Z","iopub.status.idle":"2026-06-18T06:12:11.019932Z","shell.execute_reply.started":"2026-06-18T06:12:06.695738Z","shell.execute_reply":"2026-06-18T06:12:11.019247Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Khai báo nhiều mô hình để so sánh**","metadata":{}},{"cell_type":"code","source":"models = {\n    \"Logistic Regression\": Pipeline([\n        (\"scaler\", StandardScaler()),\n        (\"model\", LogisticRegression(\n            max_iter=1000,\n            n_jobs=-1,\n            random_state=42\n        ))\n    ]),\n\n    \"Decision Tree\": DecisionTreeClassifier(\n        max_depth=20,\n        random_state=42\n    ),\n\n    \"Random Forest\": RandomForestClassifier(\n        n_estimators=100,\n        max_depth=20,\n        random_state=42,\n        n_jobs=-1\n    ),\n\n    \"Extra Trees\": ExtraTreesClassifier(\n        n_estimators=100,\n        max_depth=20,\n        random_state=42,\n        n_jobs=-1\n    ),\n\n    \"LightGBM\": LGBMClassifier(\n        n_estimators=300,\n        learning_rate=0.05,\n        num_leaves=31,\n        random_state=42,\n        n_jobs=-1\n    )\n}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-18T06:12:11.020967Z","iopub.execute_input":"2026-06-18T06:12:11.021248Z","iopub.status.idle":"2026-06-18T06:12:11.026527Z","shell.execute_reply.started":"2026-06-18T06:12:11.021226Z","shell.execute_reply":"2026-06-18T06:12:11.025622Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Huấn luyện và đánh giá mô hình**","metadata":{}},{"cell_type":"code","source":"results = []\n\nbest_model = None\nbest_model_name = None\nbest_f1 = -1\n\nfor name, model in models.items():\n    print(\"\\n\" + \"=\" * 60)\n    print(\"Đang huấn luyện mô hình:\", name)\n    print(\"=\" * 60)\n    \n    start_time = time.time()\n    \n    model.fit(X_train_small, y_train_small)\n    y_pred = model.predict(X_test_small)\n    \n    train_time = time.time() - start_time\n    \n    acc = accuracy_score(y_test_small, y_pred)\n    pre = precision_score(y_test_small, y_pred, zero_division=0)\n    rec = recall_score(y_test_small, y_pred, zero_division=0)\n    f1 = f1_score(y_test_small, y_pred, zero_division=0)\n    \n    results.append({\n        \"Model\": name,\n        \"Accuracy\": acc,\n        \"Precision\": pre,\n        \"Recall\": rec,\n        \"F1-score\": f1,\n        \"Train time (s)\": train_time\n    })\n    \n    print(\"Accuracy :\", round(acc, 4))\n    print(\"Precision:\", round(pre, 4))\n    print(\"Recall   :\", round(rec, 4))\n    print(\"F1-score :\", round(f1, 4))\n    print(\"Train time:\", round(train_time, 2), \"giây\")\n    \n    print(\"\\nClassification Report:\")\n    print(classification_report(\n        y_test_small,\n        y_pred,\n        target_names=[\"Benign\", \"Malware\"],\n        zero_division=0\n    ))\n    \n    if f1 > best_f1:\n        best_f1 = f1\n        best_model = model\n        best_model_name = name\n\nresult_df = pd.DataFrame(results)\nresult_df = result_df.sort_values(by=\"F1-score\", ascending=False)\n\nprint(\"\\nBẢNG SO SÁNH KẾT QUẢ:\")\ndisplay(result_df)\n\nprint(\"\\nMô hình tốt nhất:\", best_model_name)\nprint(\"F1-score tốt nhất:\", round(best_f1, 4))\n\njoblib.dump(best_model, \"best_ember2024_model.pkl\")\nresult_df.to_csv(\"ket_qua_so_sanh_ember2024.csv\", index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-18T06:12:11.027785Z","iopub.execute_input":"2026-06-18T06:12:11.028041Z","iopub.status.idle":"2026-06-18T06:31:06.070712Z","shell.execute_reply.started":"2026-06-18T06:12:11.028006Z","shell.execute_reply":"2026-06-18T06:31:06.069957Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Vẽ biểu đồ so sánh F1-score**","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(10, 5))\nplt.bar(result_df[\"Model\"], result_df[\"F1-score\"])\nplt.title(\"So sánh F1-score giữa các mô hình\")\nplt.xlabel(\"Mô hình\")\nplt.ylabel(\"F1-score\")\nplt.ylim(0, 1)\nplt.xticks(rotation=20)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-18T06:31:06.071815Z","iopub.execute_input":"2026-06-18T06:31:06.072111Z","iopub.status.idle":"2026-06-18T06:31:06.258947Z","shell.execute_reply.started":"2026-06-18T06:31:06.072079Z","shell.execute_reply":"2026-06-18T06:31:06.258281Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Vẽ biểu đồ so sánh Accuracy**","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(10, 5))\nplt.bar(result_df[\"Model\"], result_df[\"Accuracy\"])\nplt.title(\"So sánh Accuracy giữa các mô hình\")\nplt.xlabel(\"Mô hình\")\nplt.ylabel(\"Accuracy\")\nplt.ylim(0, 1)\nplt.xticks(rotation=20)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-18T06:31:06.260090Z","iopub.execute_input":"2026-06-18T06:31:06.260391Z","iopub.status.idle":"2026-06-18T06:31:06.369040Z","shell.execute_reply.started":"2026-06-18T06:31:06.260366Z","shell.execute_reply":"2026-06-18T06:31:06.368257Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Vẽ biểu đồ thời gian huấn luyện**","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(10, 5))\nplt.bar(result_df[\"Model\"], result_df[\"Train time (s)\"])\nplt.title(\"So sánh thời gian huấn luyện\")\nplt.xlabel(\"Mô hình\")\nplt.ylabel(\"Thời gian huấn luyện (giây)\")\nplt.xticks(rotation=20)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-18T06:31:06.369868Z","iopub.execute_input":"2026-06-18T06:31:06.370157Z","iopub.status.idle":"2026-06-18T06:31:06.479278Z","shell.execute_reply.started":"2026-06-18T06:31:06.370130Z","shell.execute_reply":"2026-06-18T06:31:06.478648Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Confusion Matrix của mô hình tốt nhất**","metadata":{}},{"cell_type":"code","source":"best_model = joblib.load(\"best_ember2024_model.pkl\")\n\ny_pred_best = best_model.predict(X_test_small)\n\ncm = confusion_matrix(y_test_small, y_pred_best)\n\ndisp = ConfusionMatrixDisplay(\n    confusion_matrix=cm,\n    display_labels=[\"Benign\", \"Malware\"]\n)\n\ndisp.plot()\nplt.title(f\"Confusion Matrix - {best_model_name}\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-18T06:31:06.480121Z","iopub.execute_input":"2026-06-18T06:31:06.480362Z","iopub.status.idle":"2026-06-18T06:31:07.216097Z","shell.execute_reply.started":"2026-06-18T06:31:06.480343Z","shell.execute_reply":"2026-06-18T06:31:07.215354Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Demo dự đoán 1 mẫu bất kỳ**","metadata":{}},{"cell_type":"code","source":"idx = random.randint(0, len(X_test_small) - 1)\n\nsample = X_test_small[idx].reshape(1, -1)\ntrue_label = int(y_test_small[idx])\n\npred = int(best_model.predict(sample)[0])\n\nlabel_name = {\n    0: \"Benign - File bình thường\",\n    1: \"Malware - File mã độc\"\n}\n\nprint(\"Mẫu số:\", idx)\nprint(\"Nhãn thật:\", label_name[true_label])\nprint(\"Dự đoán của mô hình:\", label_name[pred])\n\nif hasattr(best_model, \"predict_proba\"):\n    proba = best_model.predict_proba(sample)[0]\n    print(\"Xác suất Benign:\", round(float(proba[0]), 4))\n    print(\"Xác suất Malware:\", round(float(proba[1]), 4))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-18T06:31:07.217001Z","iopub.execute_input":"2026-06-18T06:31:07.217252Z","iopub.status.idle":"2026-06-18T06:31:07.243202Z","shell.execute_reply.started":"2026-06-18T06:31:07.217225Z","shell.execute_reply":"2026-06-18T06:31:07.242374Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Kiểm tra thêm trên Challenge Set**\n","metadata":{}},{"cell_type":"code","source":"N_CHALLENGE = 30000\n\nvalid_challenge_idx = np.where(np.isin(y_challenge, [0, 1]))[0]\n\nchallenge_idx = rng.choice(\n    valid_challenge_idx,\n    size=min(N_CHALLENGE, len(valid_challenge_idx)),\n    replace=False\n)\n\nX_challenge_small = np.array(X_challenge[challenge_idx])\ny_challenge_small = np.array(y_challenge[challenge_idx])\n\ny_pred_challenge = best_model.predict(X_challenge_small)\n\nprint(\"Kết quả trên Challenge Set:\")\nprint(classification_report(\n    y_challenge_small,\n    y_pred_challenge,\n    target_names=[\"Benign\", \"Malware\"],\n    zero_division=0\n))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-18T06:31:07.244574Z","iopub.execute_input":"2026-06-18T06:31:07.244817Z","iopub.status.idle":"2026-06-18T06:31:08.770964Z","shell.execute_reply.started":"2026-06-18T06:31:07.244775Z","shell.execute_reply":"2026-06-18T06:31:08.770293Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"Các file đã lưu:\")\nprint(\"1. best_ember2024_model.pkl\")\nprint(\"2. ket_qua_so_sanh_ember2024.csv\")\n\nprint(\"\\nBảng kết quả:\")\ndisplay(result_df)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-18T06:31:08.773620Z","iopub.execute_input":"2026-06-18T06:31:08.773850Z","iopub.status.idle":"2026-06-18T06:31:08.784435Z","shell.execute_reply.started":"2026-06-18T06:31:08.773830Z","shell.execute_reply":"2026-06-18T06:31:08.783229Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# MCVM","metadata":{}},{"cell_type":"code","source":"import os\nimport pandas as pd\nimport numpy as np\nfrom tqdm import tqdm\nfrom PIL import Image\n\nfor root, dirs, files in os.walk(\"/kaggle/input\"):\n    if \"train.7z\" in files and \"trainLabels.csv\" in files:\n        BIG_INPUT_DIR = root\n        break\n\nprint(\"BIG_INPUT_DIR:\", BIG_INPUT_DIR)\nprint(os.listdir(BIG_INPUT_DIR))\n\nTRAIN_7Z = os.path.join(BIG_INPUT_DIR, \"train.7z\")\nLABEL_PATH = os.path.join(BIG_INPUT_DIR, \"trainLabels.csv\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-19T14:52:52.835409Z","iopub.execute_input":"2026-06-19T14:52:52.835887Z","iopub.status.idle":"2026-06-19T14:52:53.206439Z","shell.execute_reply.started":"2026-06-19T14:52:52.835854Z","shell.execute_reply":"2026-06-19T14:52:53.205481Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"labels_df = pd.read_csv(LABEL_PATH)\n\nprint(labels_df.head())\nprint(labels_df[\"Class\"].value_counts().sort_index())\n\nN_PER_CLASS = 20\n\nsample_df = (\n    labels_df\n    .groupby(\"Class\", group_keys=False)\n    .apply(lambda x: x.sample(n=min(N_PER_CLASS, len(x)), random_state=42))\n    .reset_index(drop=True)\n)\n\nprint(\"Số mẫu chọn:\", len(sample_df))\nprint(sample_df[\"Class\"].value_counts().sort_index())\n\nsample_ids = sample_df[\"Id\"].tolist()\nsample_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-19T14:53:04.048852Z","iopub.execute_input":"2026-06-19T14:53:04.049296Z","iopub.status.idle":"2026-06-19T14:53:04.165334Z","shell.execute_reply.started":"2026-06-19T14:53:04.049267Z","shell.execute_reply":"2026-06-19T14:53:04.164369Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"WORK_DIR = \"/kaggle/working/big2015_mcvm_subset\"\nos.makedirs(WORK_DIR, exist_ok=True)\n\ntarget_list_path = \"/kaggle/working/targets.txt\"\n\nwith open(target_list_path, \"w\") as f:\n    for file_id in sample_ids:\n        # Thường file nằm trong thư mục train/ bên trong archive\n        f.write(f\"train/{file_id}.bytes\\n\")\n\nprint(\"Đã tạo targets.txt\")\nprint(\"Ví dụ:\")\nprint(open(target_list_path).read().splitlines()[:5])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-19T14:53:24.074898Z","iopub.execute_input":"2026-06-19T14:53:24.075245Z","iopub.status.idle":"2026-06-19T14:53:24.082997Z","shell.execute_reply.started":"2026-06-19T14:53:24.075215Z","shell.execute_reply":"2026-06-19T14:53:24.082124Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!7z x \"$TRAIN_7Z\" @\"$target_list_path\" -o\"$WORK_DIR\" -y","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-19T14:53:38.511626Z","iopub.execute_input":"2026-06-19T14:53:38.511961Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"bytes_files = []\n\nfor root, dirs, files in os.walk(WORK_DIR):\n    for file in files:\n        if file.endswith(\".bytes\"):\n            bytes_files.append(os.path.join(root, file))\n\nprint(\"Số file .bytes giải nén được:\", len(bytes_files))\nprint(bytes_files[:5])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-19T15:10:07.920009Z","iopub.execute_input":"2026-06-19T15:10:07.920331Z","iopub.status.idle":"2026-06-19T15:10:07.928600Z","shell.execute_reply.started":"2026-06-19T15:10:07.920299Z","shell.execute_reply":"2026-06-19T15:10:07.927715Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Chuyển .bytes thành ảnh grayscale**","metadata":{}},{"cell_type":"code","source":"IMG_DIR = \"/kaggle/working/mcvm_images\"\nos.makedirs(IMG_DIR, exist_ok=True)\n\ndef bytes_file_to_image(bytes_path, img_path, width=256, resize=(128, 128)):\n    values = []\n\n    with open(bytes_path, \"r\", errors=\"ignore\") as f:\n        for line in f:\n            parts = line.strip().split()\n            hex_values = parts[1:]  # bỏ địa chỉ đầu dòng\n\n            for h in hex_values:\n                if h == \"??\":\n                    values.append(0)\n                else:\n                    try:\n                        values.append(int(h, 16))\n                    except:\n                        values.append(0)\n\n    if len(values) == 0:\n        return False\n\n    arr = np.array(values, dtype=np.uint8)\n\n    height = int(np.ceil(len(arr) / width))\n    padded_len = height * width\n\n    arr = np.pad(arr, (0, padded_len - len(arr)), constant_values=0)\n    img = arr.reshape((height, width))\n\n    image = Image.fromarray(img)\n    image = image.resize(resize)\n    image.save(img_path)\n\n    return True\n\n\nimage_records = []\n\nfor bytes_path in tqdm(bytes_files):\n    file_id = os.path.basename(bytes_path).replace(\".bytes\", \"\")\n    img_path = os.path.join(IMG_DIR, file_id + \".png\")\n\n    ok = bytes_file_to_image(bytes_path, img_path)\n\n    if ok:\n        image_records.append({\n            \"Id\": file_id,\n            \"image_path\": img_path\n        })\n\nimage_df = pd.DataFrame(image_records)\n\nprint(\"Số ảnh tạo được:\", len(image_df))\nprint(image_df.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-19T15:10:12.287223Z","iopub.execute_input":"2026-06-19T15:10:12.287566Z","iopub.status.idle":"2026-06-19T15:10:51.179352Z","shell.execute_reply.started":"2026-06-19T15:10:12.287530Z","shell.execute_reply":"2026-06-19T15:10:51.178519Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Ghép ảnh với nhãn**","metadata":{}},{"cell_type":"code","source":"mcvm_df = image_df.merge(labels_df, on=\"Id\", how=\"inner\")\n\nprint(\"Số mẫu có nhãn:\", len(mcvm_df))\nprint(mcvm_df.head())\nprint(mcvm_df[\"Class\"].value_counts().sort_index())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-19T15:10:57.997046Z","iopub.execute_input":"2026-06-19T15:10:57.997359Z","iopub.status.idle":"2026-06-19T15:10:58.028728Z","shell.execute_reply.started":"2026-06-19T15:10:57.997333Z","shell.execute_reply":"2026-06-19T15:10:58.027583Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Load ảnh thành NumPy**","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import LabelEncoder\n\nIMG_SIZE = 128\n\nX = []\ny = []\n\nfor _, row in tqdm(mcvm_df.iterrows(), total=len(mcvm_df)):\n    img = Image.open(row[\"image_path\"]).convert(\"L\")\n    img = img.resize((IMG_SIZE, IMG_SIZE))\n\n    arr = np.array(img, dtype=np.float32) / 255.0\n\n    X.append(arr)\n    y.append(row[\"Class\"])\n\nX = np.array(X)\ny = np.array(y)\n\nX = X.reshape(-1, IMG_SIZE, IMG_SIZE, 1)\n\nlabel_encoder = LabelEncoder()\ny_encoded = label_encoder.fit_transform(y)\n\nprint(\"X:\", X.shape)\nprint(\"y:\", y_encoded.shape)\nprint(\"Các lớp:\", label_encoder.classes_)\nprint(\"Phân bố:\", np.unique(y_encoded, return_counts=True))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-19T15:11:03.848097Z","iopub.execute_input":"2026-06-19T15:11:03.848963Z","iopub.status.idle":"2026-06-19T15:11:05.200963Z","shell.execute_reply.started":"2026-06-19T15:11:03.848921Z","shell.execute_reply":"2026-06-19T15:11:05.199942Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Chia train/validation**","metadata":{}},{"cell_type":"code","source":"X_train_img, X_val_img, y_train_img, y_val_img = train_test_split(\n    X,\n    y_encoded,\n    test_size=0.2,\n    random_state=42,\n    stratify=y_encoded\n)\n\nprint(\"Train:\", X_train_img.shape)\nprint(\"Validation:\", X_val_img.shape)\n\nprint(\"Phân bố train:\", np.unique(y_train_img, return_counts=True))\nprint(\"Phân bố validation:\", np.unique(y_val_img, return_counts=True))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-19T15:11:14.661349Z","iopub.execute_input":"2026-06-19T15:11:14.662366Z","iopub.status.idle":"2026-06-19T15:11:14.676000Z","shell.execute_reply.started":"2026-06-19T15:11:14.662332Z","shell.execute_reply":"2026-06-19T15:11:14.675181Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Train CNN nhỏ cho MCVM","metadata":{}},{"cell_type":"code","source":"import tensorflow as tf\nfrom tensorflow.keras import layers, models\n\nnum_classes = len(np.unique(y_encoded))\n\ncnn_model = models.Sequential([\n    layers.Input(shape=(IMG_SIZE, IMG_SIZE, 1)),\n\n    layers.Conv2D(32, (3, 3), activation=\"relu\"),\n    layers.MaxPooling2D((2, 2)),\n\n    layers.Conv2D(64, (3, 3), activation=\"relu\"),\n    layers.MaxPooling2D((2, 2)),\n\n    layers.Conv2D(128, (3, 3), activation=\"relu\"),\n    layers.MaxPooling2D((2, 2)),\n\n    layers.Flatten(),\n    layers.Dense(128, activation=\"relu\"),\n    layers.Dropout(0.3),\n    layers.Dense(num_classes, activation=\"softmax\")\n])\n\ncnn_model.compile(\n    optimizer=\"adam\",\n    loss=\"sparse_categorical_crossentropy\",\n    metrics=[\"accuracy\"]\n)\n\ncnn_model.summary()\n\nhistory = cnn_model.fit(\n    X_train_img,\n    y_train_img,\n    validation_data=(X_val_img, y_val_img),\n    epochs=50,\n    batch_size=16\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-19T15:11:19.135020Z","iopub.execute_input":"2026-06-19T15:11:19.136125Z","iopub.status.idle":"2026-06-19T15:12:12.553522Z","shell.execute_reply.started":"2026-06-19T15:11:19.136082Z","shell.execute_reply":"2026-06-19T15:12:12.552619Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Đánh giá MCVM**","metadata":{}},{"cell_type":"code","source":"from sklearn.metrics import classification_report, confusion_matrix, ConfusionMatrixDisplay\nimport matplotlib.pyplot as plt\n\ny_pred_prob = cnn_model.predict(X_val_img)\ny_pred_img = np.argmax(y_pred_prob, axis=1)\n\nprint(classification_report(\n    y_val_img,\n    y_pred_img,\n    target_names=[f\"Class {c}\" for c in label_encoder.classes_],\n    zero_division=0\n))\n\ncm = confusion_matrix(y_val_img, y_pred_img)\n\ndisp = ConfusionMatrixDisplay(\n    confusion_matrix=cm,\n    display_labels=[f\"C{c}\" for c in label_encoder.classes_]\n)\n\ndisp.plot()\nplt.title(\"Confusion Matrix - MCVM CNN\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-19T15:12:40.316157Z","iopub.execute_input":"2026-06-19T15:12:40.316787Z","iopub.status.idle":"2026-06-19T15:12:41.091619Z","shell.execute_reply.started":"2026-06-19T15:12:40.316753Z","shell.execute_reply":"2026-06-19T15:12:41.090606Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Demo dự đoán một ảnh**","metadata":{}},{"cell_type":"code","source":"import random\n\nidx = random.randint(0, len(X_val_img) - 1)\n\nsample_img = X_val_img[idx]\n\ntrue_encoded = y_val_img[idx]\ntrue_class = label_encoder.inverse_transform([true_encoded])[0]\n\npred_prob = cnn_model.predict(sample_img.reshape(1, IMG_SIZE, IMG_SIZE, 1))\npred_encoded = np.argmax(pred_prob, axis=1)[0]\npred_class = label_encoder.inverse_transform([pred_encoded])[0]\n\nplt.imshow(sample_img.reshape(IMG_SIZE, IMG_SIZE), cmap=\"gray\")\nplt.title(f\"True class: {true_class} | Predicted class: {pred_class}\")\nplt.axis(\"off\")\nplt.show()\n\nprint(\"Nhãn thật:\", true_class)\nprint(\"Dự đoán:\", pred_class)\nprint(\"Xác suất cao nhất:\", round(float(np.max(pred_prob)), 4))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-19T15:12:53.149975Z","iopub.execute_input":"2026-06-19T15:12:53.150347Z","iopub.status.idle":"2026-06-19T15:12:53.413571Z","shell.execute_reply.started":"2026-06-19T15:12:53.150315Z","shell.execute_reply":"2026-06-19T15:12:53.412453Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"","metadata":{}}]}