{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":19018,"databundleVersionId":2703900,"sourceType":"competition"}],"dockerImageVersionId":31040,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-10-18T16:10:46.216198Z","iopub.execute_input":"2025-10-18T16:10:46.21651Z","iopub.status.idle":"2025-10-18T16:10:48.506487Z","shell.execute_reply.started":"2025-10-18T16:10:46.21648Z","shell.execute_reply":"2025-10-18T16:10:48.505525Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#---------------------------------------------------\n# PHẦN 1: CÀI ĐẶT VÀ CHUẨN BỊ DỮ LIỆU\n#---------------------------------------------------\n\n# 1.1. Cài đặt thư viện AutoGluon\n# Lệnh này cần thiết vì AutoGluon không có sẵn trên Kaggle\n!pip install autogluon --quiet\n\nimport pandas as pd\nfrom sklearn.model_selection import train_test_split\nfrom autogluon.tabular import TabularPredictor\n\nprint(\"=\"*50)\nprint(\"PHẦN 1: ĐANG TẢI VÀ CHUẨN BỊ DỮ LIỆU\")\nprint(\"=\"*50)\n\n# 1.2. Khai báo đường dẫn và tải dữ liệu\npath_toxic_comment = '/kaggle/input/jigsaw-multilingual-toxic-comment-classification/jigsaw-toxic-comment-train.csv'\npath_unintended_bias = '/kaggle/input/jigsaw-multilingual-toxic-comment-classification/jigsaw-unintended-bias-train.csv'\n\ntry:\n    df_toxic = pd.read_csv(path_toxic_comment)\n    df_bias = pd.read_csv(path_unintended_bias)\n\n    # Ghép nối và chuẩn hóa dữ liệu như các bước trước\n    df_toxic_subset = df_toxic[['comment_text', 'toxic']]\n    df_bias_subset = df_bias[['comment_text', 'toxic']]\n    full_train_df = pd.concat([df_toxic_subset, df_bias_subset], ignore_index=True)\n    full_train_df['toxic'] = full_train_df['toxic'].apply(lambda x: 1 if x >= 0.5 else 0)\n    \n    # Để chạy nhanh hơn cho ví dụ này, chúng ta sẽ lấy một mẫu nhỏ\n    # BỎ CHÚ THÍCH DÒNG DƯỚI ĐÂY NẾU BẠN MUỐN CHẠY TRÊN TOÀN BỘ DỮ LIỆU (sẽ mất rất nhiều thời gian)\n    full_train_df = full_train_df.sample(n=100000, random_state=42)\n    \n    print(f\"Đã tạo DataFrame training tổng hợp với {len(full_train_df)} mẫu.\")\n\nexcept FileNotFoundError as e:\n    print(f\"\\nLỖI: Không tìm thấy file training. Quy trình dừng lại.\")\n    print(f\"Chi tiết lỗi: {e}\")\n    exit()\n\n#---------------------------------------------------\n# PHẦN 2: CHIA DỮ LIỆU (80% TRAIN, 20% TEST)\n#---------------------------------------------------\nprint(\"\\n\" + \"=\"*50)\nprint(\"PHẦN 2: CHIA DỮ LIỆU THÀNH TẬP TRAIN VÀ TEST\")\nprint(\"=\"*50)\n\n# Chia dữ liệu thành 80% train và 20% test (để đánh giá cuối cùng)\n# stratify=full_train_df['toxic'] rất quan trọng để đảm bảo tỉ lệ nhãn 'toxic'\n# là như nhau trong cả hai tập train và test.\ntrain_data, test_data = train_test_split(\n    full_train_df,\n    test_size=0.2,\n    random_state=42,\n    stratify=full_train_df['toxic']\n)\n\nprint(f\"Kích thước tập Train: {train_data.shape}\")\nprint(f\"Kích thước tập Test: {test_data.shape}\")\nprint(f\"Phân phối nhãn trong tập Train:\\n{train_data['toxic'].value_counts(normalize=True)}\")\nprint(f\"Phân phối nhãn trong tập Test:\\n{test_data['toxic'].value_counts(normalize=True)}\")\n\n\n#---------------------------------------------------\n# PHẦN 3: HUẤN LUYỆN VỚI AUTOGLUON\n#---------------------------------------------------\nprint(\"\\n\" + \"=\"*50)\nprint(\"PHẦN 3: BẮT ĐẦU HUẤN LUYỆN VỚI AUTOGLUON\")\nprint(\"=\"*50)\n\n# Khởi tạo TabularPredictor\n# AutoGluon sẽ tự động xử lý cột 'comment_text' như một đặc trưng văn bản\npredictor = TabularPredictor(\n    label='toxic',                # Cột mục tiêu cần dự đoán\n    problem_type='binary',        # Loại bài toán: phân loại nhị phân\n    eval_metric='roc_auc',        # Thước đo để tối ưu, phù hợp với cuộc thi\n    path='./ag_models_toxic'      # Thư mục để lưu các mô hình đã huấn luyện\n)\n\n# Huấn luyện mô hình\n# AutoGluon sẽ thử nhiều mô hình khác nhau và kết hợp chúng lại\n# time_limit là giới hạn thời gian huấn luyện (tính bằng giây)\n# presets='best_quality' để có kết quả tốt nhất, bạn có thể dùng 'high_quality' hoặc 'medium_quality' để nhanh hơn\npredictor.fit(\n    train_data,\n    time_limit=1800, # Giới hạn thời gian 30 phút. Tăng lên để có kết quả tốt hơn.\n    presets='high_quality'\n)\n\n#---------------------------------------------------\n# PHẦN 4: ĐÁNH GIÁ MÔ HÌNH TRÊN TẬP TEST\n#---------------------------------------------------\nprint(\"\\n\" + \"=\"*50)\nprint(\"PHẦN 4: ĐÁNH GIÁ HIỆU SUẤT MÔ HÌNH\")\nprint(\"=\"*50)\n\n# Xem bảng xếp hạng các mô hình đã được huấn luyện\nprint(\"Bảng xếp hạng các mô hình (đánh giá trên tập validation nội bộ của AutoGluon):\")\nleaderboard = predictor.leaderboard(silent=True)\nprint(leaderboard)\n\n# Đánh giá hiệu suất trên tập test 20% mà chúng ta đã tách ra\nprint(\"\\nĐánh giá trên tập test (20% dữ liệu giữ lại):\")\nperformance = predictor.evaluate(test_data)\nprint(performance)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-18T16:10:54.155436Z","iopub.execute_input":"2025-10-18T16:10:54.155813Z","iopub.status.idle":"2025-10-18T16:30:32.642104Z","shell.execute_reply.started":"2025-10-18T16:10:54.155786Z","shell.execute_reply":"2025-10-18T16:30:32.64011Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Cài đặt các thư viện cần thiết\n!pip install -q transformers datasets torch\n\nimport pandas as pd\nimport numpy as np\nimport torch\nfrom torch.utils.data import Dataset, DataLoader\nfrom torch import nn\nfrom transformers import AutoTokenizer, AutoModel, AdamW\nfrom sklearn.model_selection import train_test_split\nfrom tqdm.notebook import tqdm\nimport os\n\n# Đảm bảo kết quả có thể tái tạo\ndef set_seed(seed=42):\n    np.random.seed(seed)\n    torch.manual_seed(seed)\n    torch.cuda.manual_seed_all(seed)\n    torch.backends.cudnn.deterministic = True\n    torch.backends.cudnn.benchmark = False\n\nset_seed(42)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-18T16:38:11.537391Z","iopub.execute_input":"2025-10-18T16:38:11.540294Z","iopub.status.idle":"2025-10-18T16:38:25.810523Z","shell.execute_reply.started":"2025-10-18T16:38:11.540232Z","shell.execute_reply":"2025-10-18T16:38:25.809418Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class Config:\n    MODEL_NAME = 'xlm-roberta-base' # Mô hình đa ngôn ngữ mạnh mẽ\n    MAX_LEN = 192                  # Độ dài tối đa của chuỗi đầu vào\n    BATCH_SIZE = 16                # Giảm nếu gặp lỗi Out-of-Memory (OOM)\n    EPOCHS = 1                     # Fine-tuning 1-2 epochs thường là đủ và tránh overfitting\n    LEARNING_RATE = 3e-5           # Tốc độ học phổ biến cho fine-tuning\n    DEVICE = torch.device('cuda' if torch.cuda.is_available() else 'cpu')\n\n    # Đường dẫn file trên Kaggle\n    PATH = '/kaggle/input/jigsaw-multilingual-toxic-comment-classification/'\n\nconfig = Config()\nprint(f\"Sử dụng thiết bị: {config.DEVICE}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-18T16:39:35.144243Z","iopub.execute_input":"2025-10-18T16:39:35.146163Z","iopub.status.idle":"2025-10-18T16:39:35.154017Z","shell.execute_reply.started":"2025-10-18T16:39:35.14612Z","shell.execute_reply":"2025-10-18T16:39:35.152863Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Tải dữ liệu\ndf_train = pd.read_csv(config.PATH + 'jigsaw-toxic-comment-train.csv')\ndf_valid = pd.read_csv(config.PATH + 'validation.csv')\ndf_test = pd.read_csv(config.PATH + 'test.csv')\n\n\n\n# Các cột nhãn có trong tập train\nTRAIN_LABEL_COLUMNS = ['toxic', 'severe_toxic', 'obscene', 'threat', 'insult', 'identity_hate']\n\n# ... (tải df_train, df_valid, df_test)\n\n# SỬA LẠI LỚP DATASET Ở ĐÂY\nclass JigsawDataset(Dataset):\n    def __init__(self, df, tokenizer, max_len, is_test=False):\n        self.df = df\n        self.tokenizer = tokenizer\n        self.max_len = max_len\n        self.is_test = is_test\n        \n        # Lấy cột văn bản (có tên khác nhau trong các file)\n        if 'comment_text' in df.columns:\n            self.texts = df['comment_text'].values\n        else:\n            self.texts = df['content'].values\n        \n        if not is_test:\n            # ---> LOGIC SỬA LỖI <---\n            # Tìm các cột nhãn có trong DataFrame hiện tại\n            self.target_cols = [col for col in TRAIN_LABEL_COLUMNS if col in df.columns]\n            self.labels = df[self.target_cols].values\n\n    def __len__(self):\n        return len(self.df)\n\n    def __getitem__(self, idx):\n        text = str(self.texts[idx])\n        inputs = self.tokenizer.encode_plus(\n            text,\n            add_special_tokens=True,\n            max_length=self.max_len,\n            padding='max_length',\n            truncation=True,\n            return_token_type_ids=False,\n            return_tensors='pt'\n        )\n        \n        ids = inputs['input_ids'].squeeze(0)\n        mask = inputs['attention_mask'].squeeze(0)\n\n        if self.is_test:\n            return {'ids': ids, 'mask': mask}\n        else:\n            # Logic này sẽ trả về 6 nhãn cho tập train và 1 nhãn cho tập valid\n            targets = torch.tensor(self.labels[idx], dtype=torch.float)\n            return {'ids': ids, 'mask': mask, 'targets': targets}\n\n# KHI TẠO DATASET, MỌI THỨ VẪN NHƯ CŨ\ntrain_dataset = JigsawDataset(df_train, tokenizer, config.MAX_LEN)\nvalid_dataset = JigsawDataset(df_valid, tokenizer, config.MAX_LEN)\ntest_dataset = JigsawDataset(df_test, tokenizer, config.MAX_LEN, is_test=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-18T16:43:45.959339Z","iopub.execute_input":"2025-10-18T16:43:45.960491Z","iopub.status.idle":"2025-10-18T16:43:48.887111Z","shell.execute_reply.started":"2025-10-18T16:43:45.960452Z","shell.execute_reply":"2025-10-18T16:43:48.886138Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-18T16:40:15.966899Z","iopub.execute_input":"2025-10-18T16:40:15.967308Z","iopub.status.idle":"2025-10-18T16:40:15.989997Z","shell.execute_reply.started":"2025-10-18T16:40:15.96728Z","shell.execute_reply":"2025-10-18T16:40:15.988881Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class JigsawModel(nn.Module):\n    def __init__(self, model_name, num_labels):\n        super().__init__()\n        self.bert = AutoModel.from_pretrained(model_name)\n        self.dropout = nn.Dropout(0.1)\n        self.classifier = nn.Linear(self.bert.config.hidden_size, num_labels)\n\n    def forward(self, ids, mask):\n        # Lấy output của token [CLS] (pooled output)\n        _, pooled_output = self.bert(input_ids=ids, attention_mask=mask, return_dict=False)\n        output = self.dropout(pooled_output)\n        logits = self.classifier(output)\n        return logits\n\n# Khởi tạo mô hình\nmodel = JigsawModel(config.MODEL_NAME, len(LABEL_COLUMNS))\nmodel.to(config.DEVICE);","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-18T16:44:03.185157Z","iopub.execute_input":"2025-10-18T16:44:03.185775Z","iopub.status.idle":"2025-10-18T16:44:31.287662Z","shell.execute_reply.started":"2025-10-18T16:44:03.185745Z","shell.execute_reply":"2025-10-18T16:44:31.286436Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_loader = DataLoader(train_dataset, batch_size=config.BATCH_SIZE, shuffle=True)\ntest_loader = DataLoader(test_dataset, batch_size=config.BATCH_SIZE, shuffle=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-18T16:45:33.80729Z","iopub.execute_input":"2025-10-18T16:45:33.807683Z","iopub.status.idle":"2025-10-18T16:45:33.813607Z","shell.execute_reply.started":"2025-10-18T16:45:33.807658Z","shell.execute_reply":"2025-10-18T16:45:33.812692Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Hàm mất mát và optimizer\nloss_fn = nn.BCEWithLogitsLoss() # Phù hợp cho multi-label classification\noptimizer = AdamW(model.parameters(), lr=config.LEARNING_RATE)\n\n# Vòng lặp huấn luyện\nprint(\"Bắt đầu fine-tuning...\")\nfor epoch in range(config.EPOCHS):\n    model.train()\n    total_loss = 0\n    for batch in tqdm(train_loader, desc=f\"Training Epoch {epoch + 1}\"):\n        ids = batch['ids'].to(config.DEVICE)\n        mask = batch['mask'].to(config.DEVICE)\n        targets = batch['targets'].to(config.DEVICE)\n\n        optimizer.zero_grad()\n        outputs = model(ids=ids, mask=mask)\n        loss = loss_fn(outputs, targets)\n        \n        total_loss += loss.item()\n        loss.backward()\n        optimizer.step()\n    \n    avg_loss = total_loss / len(train_loader)\n    print(f\"Epoch {epoch + 1}/{config.EPOCHS}, Average Training Loss: {avg_loss:.4f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-18T16:45:36.913703Z","iopub.execute_input":"2025-10-18T16:45:36.914096Z","iopub.status.idle":"2025-10-18T16:49:26.76526Z","shell.execute_reply.started":"2025-10-18T16:45:36.914069Z","shell.execute_reply":"2025-10-18T16:49:26.763926Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def prediction_fn(data_loader, model, device):\n    model.eval()\n    final_outputs = []\n    with torch.no_grad():\n        for batch in tqdm(data_loader, desc=\"Predicting\"):\n            ids = batch['ids'].to(device)\n            mask = batch['mask'].to(device)\n            \n            outputs = model(ids=ids, mask=mask)\n            # Dùng sigmoid để chuyển logits thành xác suất từ 0 đến 1\n            final_outputs.extend(torch.sigmoid(outputs).cpu().detach().numpy().tolist())\n    return final_outputs\n\npredictions = prediction_fn(test_loader, model, config.DEVICE)\n\n# Tạo file submission\nsubmission_df = pd.DataFrame(predictions, columns=LABEL_COLUMNS)\nsubmission_df['id'] = df_test['id']\nsubmission_df = submission_df[['id'] + LABEL_COLUMNS]\nsubmission_df.to_csv('submission.csv', index=False)\n\nprint(\"\\nĐã tạo file submission.csv thành công!\")\nprint(submission_df.head())","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}