{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"codemirror_mode":{"name":"ipython","version":3},"file_extension":".py","mimetype":"text/x-python","name":"python","nbconvert_exporter":"python","pygments_lexer":"ipython3","version":"3.12.12"},"kaggle":{"accelerator":"none","dataSources":[{"databundleVersionId":46665,"sourceId":4117,"sourceType":"competition"},{"sourceId":309915035,"sourceType":"kernelVersion"},{"sourceId":310094652,"sourceType":"kernelVersion"}],"dockerImageVersionId":31328,"isGpuEnabled":false,"isInternetEnabled":true,"language":"python","sourceType":"notebook"},"papermill":{"default_parameters":{},"duration":4.974675,"end_time":"2026-04-09T03:04:09.126601+00:00","environment_variables":{},"exception":null,"input_path":"__notebook__.ipynb","output_path":"__notebook__.ipynb","parameters":{},"start_time":"2026-04-09T03:04:04.151926+00:00","version":"2.7.0"}},"nbformat_minor":4,"nbformat":4,"cells":[{"id":"fb27296c","cell_type":"code","source":"import pandas as pd\nimport numpy as np\n\n# ==========================================\n# 1. CẤU HÌNH ĐƯỜNG DẪN VÀ TRỌNG SỐ\n# ==========================================\n# Đường dẫn tới 2 file nộp bài tốt nhất của bạn\nFILE_XGB = '/kaggle/input/notebooks/nguynhongdiumy/meta-stacking-tree-based-submit/submission_malware_2015.csv' # File điểm 0.02323\nFILE_DL = '/kaggle/input/notebooks/huylhn1810/dual-band-swin-tiny-resnet18-ver-1/submission.csv' # File điểm 0.02795\n\nOUTPUT_BLEND = '/kaggle/working/submission_blended.csv'\n\n# Cài đặt trọng số (XGBoost đang tốt hơn nên cho chiếm tỷ trọng cao hơn)\n# Bạn có thể tinh chỉnh cặp số này (ví dụ: 0.6 - 0.4, hoặc 0.8 - 0.2)\nWEIGHT_XGB = 0.45\nWEIGHT_DL = 0.55\n\nassert abs(WEIGHT_XGB + WEIGHT_DL - 1.0) < 1e-5, \"Tổng trọng số phải bằng 1.0!\"\n\ndef soft_blending():\n    print(\"Đang đọc dữ liệu dự đoán...\")\n    df_xgb = pd.read_csv(FILE_XGB)\n    df_dl = pd.read_csv(FILE_DL)\n    \n    # BƯỚC QUAN TRỌNG: Đảm bảo 2 file được sắp xếp cùng thứ tự ID trước khi cộng\n    # Nếu ID lệch nhau, việc cộng xác suất sẽ là thảm họa!\n    df_xgb = df_xgb.sort_values(by='Id').reset_index(drop=True)\n    df_dl = df_dl.sort_values(by='Id').reset_index(drop=True)\n    \n    # Kiểm tra an toàn: Đảm bảo tập ID khớp nhau 100%\n    if not df_xgb['Id'].equals(df_dl['Id']):\n        raise ValueError(\"LỖI: Cột 'Id' của 2 file không khớp nhau. Hãy kiểm tra lại!\")\n        \n    print(f\"Đã xác nhận ID khớp nhau hoàn toàn. Tổng cộng {len(df_xgb)} tệp.\")\n    \n    # Bắt đầu trộn xác suất\n    print(f\"Đang trộn xác suất với tỷ lệ: XGBoost ({WEIGHT_XGB}) + Deep Learning ({WEIGHT_DL})\")\n    \n    # Tạo một DataFrame mới copy từ df_xgb để giữ nguyên cột Id\n    df_blend = df_xgb.copy()\n    \n    # Các cột chứa xác suất từ Prediction1 đến Prediction9\n    pred_cols = [f'Prediction{i}' for i in range(1, 10)]\n    \n    # Tính trung bình cộng có trọng số cho từng cột\n    for col in pred_cols:\n        df_blend[col] = (df_xgb[col] * WEIGHT_XGB) + (df_dl[col] * WEIGHT_DL)\n        \n    # Lưu file kết quả\n    df_blend.to_csv(OUTPUT_BLEND, index=False)\n    print(f\"HOÀN TẤT! File nộp bài đã được lưu tại: {OUTPUT_BLEND}\")\n\n# Chạy chương trình\nif __name__ == '__main__':\n    soft_blending()","metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","execution":{"iopub.execute_input":"2026-04-09T03:04:07.257948Z","iopub.status.busy":"2026-04-09T03:04:07.257617Z","iopub.status.idle":"2026-04-09T03:04:08.702983Z","shell.execute_reply":"2026-04-09T03:04:08.701976Z"},"papermill":{"duration":1.450309,"end_time":"2026-04-09T03:04:08.704542+00:00","exception":false,"start_time":"2026-04-09T03:04:07.254233+00:00","status":"completed"},"tags":[]},"outputs":[],"execution_count":null}]}