{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":19018,"databundleVersionId":2703900,"sourceType":"competition"}],"dockerImageVersionId":31040,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-06-29T16:41:29.777494Z","iopub.execute_input":"2025-06-29T16:41:29.777686Z","iopub.status.idle":"2025-06-29T16:41:31.308553Z","shell.execute_reply.started":"2025-06-29T16:41:29.777668Z","shell.execute_reply":"2025-06-29T16:41:31.307559Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#---------------------------------------------------\n# PHẦN 1: CÀI ĐẶT VÀ CHUẨN BỊ DỮ LIỆU\n#---------------------------------------------------\n\n# 1.1. Cài đặt thư viện AutoGluon\n# Lệnh này cần thiết vì AutoGluon không có sẵn trên Kaggle\n!pip install autogluon --quiet\n\nimport pandas as pd\nfrom sklearn.model_selection import train_test_split\nfrom autogluon.tabular import TabularPredictor\n\nprint(\"=\"*50)\nprint(\"PHẦN 1: ĐANG TẢI VÀ CHUẨN BỊ DỮ LIỆU\")\nprint(\"=\"*50)\n\n# 1.2. Khai báo đường dẫn và tải dữ liệu\npath_toxic_comment = '/kaggle/input/jigsaw-multilingual-toxic-comment-classification/jigsaw-toxic-comment-train.csv'\npath_unintended_bias = '/kaggle/input/jigsaw-multilingual-toxic-comment-classification/jigsaw-unintended-bias-train.csv'\n\ntry:\n    df_toxic = pd.read_csv(path_toxic_comment)\n    df_bias = pd.read_csv(path_unintended_bias)\n\n    # Ghép nối và chuẩn hóa dữ liệu như các bước trước\n    df_toxic_subset = df_toxic[['comment_text', 'toxic']]\n    df_bias_subset = df_bias[['comment_text', 'toxic']]\n    full_train_df = pd.concat([df_toxic_subset, df_bias_subset], ignore_index=True)\n    full_train_df['toxic'] = full_train_df['toxic'].apply(lambda x: 1 if x >= 0.5 else 0)\n    \n    # Để chạy nhanh hơn cho ví dụ này, chúng ta sẽ lấy một mẫu nhỏ\n    # BỎ CHÚ THÍCH DÒNG DƯỚI ĐÂY NẾU BẠN MUỐN CHẠY TRÊN TOÀN BỘ DỮ LIỆU (sẽ mất rất nhiều thời gian)\n    full_train_df = full_train_df.sample(n=100000, random_state=42)\n    \n    print(f\"Đã tạo DataFrame training tổng hợp với {len(full_train_df)} mẫu.\")\n\nexcept FileNotFoundError as e:\n    print(f\"\\nLỖI: Không tìm thấy file training. Quy trình dừng lại.\")\n    print(f\"Chi tiết lỗi: {e}\")\n    exit()\n\n#---------------------------------------------------\n# PHẦN 2: CHIA DỮ LIỆU (80% TRAIN, 20% TEST)\n#---------------------------------------------------\nprint(\"\\n\" + \"=\"*50)\nprint(\"PHẦN 2: CHIA DỮ LIỆU THÀNH TẬP TRAIN VÀ TEST\")\nprint(\"=\"*50)\n\n# Chia dữ liệu thành 80% train và 20% test (để đánh giá cuối cùng)\n# stratify=full_train_df['toxic'] rất quan trọng để đảm bảo tỉ lệ nhãn 'toxic'\n# là như nhau trong cả hai tập train và test.\ntrain_data, test_data = train_test_split(\n    full_train_df,\n    test_size=0.2,\n    random_state=42,\n    stratify=full_train_df['toxic']\n)\n\nprint(f\"Kích thước tập Train: {train_data.shape}\")\nprint(f\"Kích thước tập Test: {test_data.shape}\")\nprint(f\"Phân phối nhãn trong tập Train:\\n{train_data['toxic'].value_counts(normalize=True)}\")\nprint(f\"Phân phối nhãn trong tập Test:\\n{test_data['toxic'].value_counts(normalize=True)}\")\n\n\n#---------------------------------------------------\n# PHẦN 3: HUẤN LUYỆN VỚI AUTOGLUON\n#---------------------------------------------------\nprint(\"\\n\" + \"=\"*50)\nprint(\"PHẦN 3: BẮT ĐẦU HUẤN LUYỆN VỚI AUTOGLUON\")\nprint(\"=\"*50)\n\n# Khởi tạo TabularPredictor\n# AutoGluon sẽ tự động xử lý cột 'comment_text' như một đặc trưng văn bản\npredictor = TabularPredictor(\n    label='toxic',                # Cột mục tiêu cần dự đoán\n    problem_type='binary',        # Loại bài toán: phân loại nhị phân\n    eval_metric='roc_auc',        # Thước đo để tối ưu, phù hợp với cuộc thi\n    path='./ag_models_toxic'      # Thư mục để lưu các mô hình đã huấn luyện\n)\n\n# Huấn luyện mô hình\n# AutoGluon sẽ thử nhiều mô hình khác nhau và kết hợp chúng lại\n# time_limit là giới hạn thời gian huấn luyện (tính bằng giây)\n# presets='best_quality' để có kết quả tốt nhất, bạn có thể dùng 'high_quality' hoặc 'medium_quality' để nhanh hơn\npredictor.fit(\n    train_data,\n    time_limit=1800, # Giới hạn thời gian 30 phút. Tăng lên để có kết quả tốt hơn.\n    presets='high_quality'\n)\n\n#---------------------------------------------------\n# PHẦN 4: ĐÁNH GIÁ MÔ HÌNH TRÊN TẬP TEST\n#---------------------------------------------------\nprint(\"\\n\" + \"=\"*50)\nprint(\"PHẦN 4: ĐÁNH GIÁ HIỆU SUẤT MÔ HÌNH\")\nprint(\"=\"*50)\n\n# Xem bảng xếp hạng các mô hình đã được huấn luyện\nprint(\"Bảng xếp hạng các mô hình (đánh giá trên tập validation nội bộ của AutoGluon):\")\nleaderboard = predictor.leaderboard(silent=True)\nprint(leaderboard)\n\n# Đánh giá hiệu suất trên tập test 20% mà chúng ta đã tách ra\nprint(\"\\nĐánh giá trên tập test (20% dữ liệu giữ lại):\")\nperformance = predictor.evaluate(test_data)\nprint(performance)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-29T16:41:31.310314Z","iopub.execute_input":"2025-06-29T16:41:31.310829Z","iopub.status.idle":"2025-06-29T17:22:51.982697Z","shell.execute_reply.started":"2025-06-29T16:41:31.310798Z","shell.execute_reply":"2025-06-29T17:22:51.981596Z"}},"outputs":[],"execution_count":null}]}