{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":10737,"databundleVersionId":290346,"sourceType":"competition"}],"dockerImageVersionId":31012,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Import các thư viện cần thiết\nimport pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom collections import Counter\nimport re\nimport nltk\nfrom nltk.corpus import stopwords\nfrom wordcloud import WordCloud\nimport os\nimport warnings\nwarnings.filterwarnings('ignore')\n\n# Thiết lập hiển thị đồ thị\nplt.style.use('ggplot')\nsns.set(style='whitegrid')\n%matplotlib inline","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-10T01:53:32.801756Z","iopub.execute_input":"2025-05-10T01:53:32.802767Z","iopub.status.idle":"2025-05-10T01:53:32.810307Z","shell.execute_reply.started":"2025-05-10T01:53:32.802720Z","shell.execute_reply":"2025-05-10T01:53:32.809199Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Đường dẫn đến dữ liệu trên Kaggle\ndata_path = \"../input/quora-insincere-questions-classification/\"\n\n# Kiểm tra các file có sẵn\nprint(\"Các file có sẵn:\")\nfor file in os.listdir(data_path):\n    print(f\"- {file}\")\n\n# Đọc dữ liệu\ntrain_df = pd.read_csv(f\"{data_path}train.csv\")\ntest_df = pd.read_csv(f\"{data_path}test.csv\")\n\n# Tải các tài nguyên NLTK cần thiết\nnltk.download('stopwords')\nstop_words = set(stopwords.words('english'))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-10T01:54:03.030433Z","iopub.execute_input":"2025-05-10T01:54:03.030753Z","iopub.status.idle":"2025-05-10T01:54:07.350346Z","shell.execute_reply.started":"2025-05-10T01:54:03.030731Z","shell.execute_reply":"2025-05-10T01:54:07.349378Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Xem kích thước của dữ liệu\nprint(f\"Kích thước dữ liệu train: {train_df.shape}\")\nprint(f\"Kích thước dữ liệu test: {test_df.shape}\")\n\n# Hiển thị một vài dòng đầu tiên\nprint(\"\\nMột vài dòng đầu tiên trong tập train:\")\ndisplay(train_df.head())\n\n# Kiểm tra thông tin cột\nprint(\"\\nThông tin về các cột trong tập train:\")\ntrain_df.info()\n\n# Kiểm tra các giá trị bị thiếu\nprint(\"\\nSố lượng giá trị bị thiếu trong tập train:\")\nprint(train_df.isnull().sum())\n\n# Kiểm tra phân phối nhãn\nprint(\"\\nPhân phối nhãn trong tập train:\")\nlabel_dist = train_df['target'].value_counts()\nprint(label_dist)\nprint(f\"Tỷ lệ câu hỏi không chân thành: {train_df['target'].mean()*100:.2f}%\")\n\n# Vẽ biểu đồ phân phối nhãn\nplt.figure(figsize=(8, 5))\nsns.countplot(x='target', data=train_df)\nplt.title('Phân phối nhãn trong tập train')\nplt.xlabel('Nhãn (0: Chân thành, 1: Không chân thành)')\nplt.ylabel('Số lượng')\nfor i, count in enumerate(label_dist.values):\n    plt.text(i, count + 1000, f\"{count} ({count/len(train_df)*100:.2f}%)\", ha='center')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-10T01:54:33.745228Z","iopub.execute_input":"2025-05-10T01:54:33.746030Z","iopub.status.idle":"2025-05-10T01:54:34.347643Z","shell.execute_reply.started":"2025-05-10T01:54:33.745996Z","shell.execute_reply":"2025-05-10T01:54:34.346550Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Thêm các cột độ dài\ntrain_df['char_length'] = train_df['question_text'].apply(len)\ntrain_df['word_count'] = train_df['question_text'].apply(lambda x: len(str(x).split()))\n\n# Tính toán thống kê cho độ dài\nprint(\"Thống kê về độ dài câu hỏi theo nhãn:\")\nlength_stats = train_df.groupby('target')[['char_length', 'word_count']].agg(['mean', 'median', 'min', 'max']).round(2)\ndisplay(length_stats)\n\n# Vẽ biểu đồ phân phối độ dài\nfig, ax = plt.subplots(1, 2, figsize=(16, 6))\n\n# Biểu đồ phân phối số ký tự\nsns.histplot(data=train_df, x='char_length', hue='target', bins=50, kde=True, ax=ax[0])\nax[0].set_title('Phân phối độ dài câu hỏi (số ký tự)')\nax[0].set_xlabel('Số ký tự')\nax[0].set_ylabel('Số lượng câu hỏi')\nax[0].set_xlim(0, 300)\n\n# Biểu đồ phân phối số từ\nsns.histplot(data=train_df, x='word_count', hue='target', bins=50, kde=True, ax=ax[1])\nax[1].set_title('Phân phối độ dài câu hỏi (số từ)')\nax[1].set_xlabel('Số từ')\nax[1].set_ylabel('Số lượng câu hỏi')\nax[1].set_xlim(0, 50)\n\nplt.tight_layout()\nplt.show()\n\n# Boxplot so sánh độ dài theo nhãn\nfig, ax = plt.subplots(1, 2, figsize=(16, 6))\n\nsns.boxplot(x='target', y='char_length', data=train_df, ax=ax[0])\nax[0].set_title('So sánh độ dài câu hỏi (số ký tự) theo nhãn')\nax[0].set_xlabel('Nhãn (0: Chân thành, 1: Không chân thành)')\nax[0].set_ylabel('Số ký tự')\n\nsns.boxplot(x='target', y='word_count', data=train_df, ax=ax[1])\nax[1].set_title('So sánh độ dài câu hỏi (số từ) theo nhãn')\nax[1].set_xlabel('Nhãn (0: Chân thành, 1: Không chân thành)')\nax[1].set_ylabel('Số từ')\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-10T01:54:52.225600Z","iopub.execute_input":"2025-05-10T01:54:52.225989Z","iopub.status.idle":"2025-05-10T01:55:07.972412Z","shell.execute_reply.started":"2025-05-10T01:54:52.225954Z","shell.execute_reply":"2025-05-10T01:55:07.971417Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Thêm các đặc trưng văn bản\ntrain_df['question_marks'] = train_df['question_text'].apply(lambda x: x.count('?'))\ntrain_df['exclamation_marks'] = train_df['question_text'].apply(lambda x: x.count('!'))\ntrain_df['uppercase_count'] = train_df['question_text'].apply(lambda x: sum(1 for c in x if c.isupper()))\ntrain_df['uppercase_ratio'] = train_df['uppercase_count'] / train_df['char_length'].apply(lambda x: max(x, 1))\n\n# Vẽ biểu đồ so sánh các đặc trưng văn bản theo nhãn\nfig, axes = plt.subplots(2, 2, figsize=(16, 12))\n\nsns.boxplot(x='target', y='question_marks', data=train_df, ax=axes[0, 0])\naxes[0, 0].set_title('Số lượng dấu hỏi theo nhãn')\naxes[0, 0].set_xlabel('Nhãn (0: Chân thành, 1: Không chân thành)')\n\nsns.boxplot(x='target', y='exclamation_marks', data=train_df, ax=axes[0, 1])\naxes[0, 1].set_title('Số lượng dấu chấm than theo nhãn')\naxes[0, 1].set_xlabel('Nhãn (0: Chân thành, 1: Không chân thành)')\n\nsns.boxplot(x='target', y='uppercase_count', data=train_df, ax=axes[1, 0])\naxes[1, 0].set_title('Số lượng chữ in hoa theo nhãn')\naxes[1, 0].set_xlabel('Nhãn (0: Chân thành, 1: Không chân thành)')\n\nsns.boxplot(x='target', y='uppercase_ratio', data=train_df, ax=axes[1, 1])\naxes[1, 1].set_title('Tỷ lệ chữ in hoa theo nhãn')\naxes[1, 1].set_xlabel('Nhãn (0: Chân thành, 1: Không chân thành)')\n\nplt.tight_layout()\nplt.show()\n\n# Thống kê đặc trưng văn bản theo nhãn\nfeatures_stats = train_df.groupby('target')[['question_marks', 'exclamation_marks', 'uppercase_count', 'uppercase_ratio']].agg(['mean', 'median']).round(3)\nprint(\"Thống kê các đặc trưng văn bản theo nhãn:\")\ndisplay(features_stats)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-10T01:55:18.856434Z","iopub.execute_input":"2025-05-10T01:55:18.856740Z","iopub.status.idle":"2025-05-10T01:55:26.337248Z","shell.execute_reply.started":"2025-05-10T01:55:18.856720Z","shell.execute_reply":"2025-05-10T01:55:26.336403Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Hàm tiền xử lý văn bản\ndef preprocess_text(text):\n    # Chuyển về chữ thường\n    text = text.lower()\n    # Loại bỏ ký tự đặc biệt\n    text = re.sub(r'[^\\w\\s]', '', text)\n    # Tách từ\n    words = text.split()\n    # Loại bỏ stopwords\n    words = [word for word in words if word not in stop_words]\n    return words\n\n# Phân tách dữ liệu theo nhãn\nsincere_questions = train_df[train_df['target'] == 0]['question_text']\ninsincere_questions = train_df[train_df['target'] == 1]['question_text']\n\n# Lấy tất cả các từ\nsincere_words = []\nfor question in sincere_questions:\n    sincere_words.extend(preprocess_text(question))\n    \ninsincere_words = []\nfor question in insincere_questions:\n    insincere_words.extend(preprocess_text(question))\n\n# Đếm các từ phổ biến\nsincere_word_counts = Counter(sincere_words).most_common(20)\ninsincere_word_counts = Counter(insincere_words).most_common(20)\n\n# Vẽ biểu đồ các từ phổ biến\nfig, ax = plt.subplots(1, 2, figsize=(20, 8))\n\n# Từ phổ biến trong câu hỏi chân thành\nsincere_df = pd.DataFrame(sincere_word_counts, columns=['word', 'count'])\nsns.barplot(x='count', y='word', data=sincere_df, ax=ax[0])\nax[0].set_title('20 từ phổ biến nhất trong câu hỏi chân thành')\nax[0].set_xlabel('Số lần xuất hiện')\n\n# Từ phổ biến trong câu hỏi không chân thành\ninsincere_df = pd.DataFrame(insincere_word_counts, columns=['word', 'count'])\nsns.barplot(x='count', y='word', data=insincere_df, ax=ax[1])\nax[1].set_title('20 từ phổ biến nhất trong câu hỏi không chân thành')\nax[1].set_xlabel('Số lần xuất hiện')\n\nplt.tight_layout()\nplt.show()\n\n# Tạo WordCloud\nfig, ax = plt.subplots(1, 2, figsize=(20, 10))\n\n# WordCloud cho câu hỏi chân thành\nsincere_text = ' '.join(sincere_words)\nwordcloud_sincere = WordCloud(width=800, height=400, background_color='white', \n                             max_words=200, contour_width=3, contour_color='steelblue')\nwordcloud_sincere.generate(sincere_text)\nax[0].imshow(wordcloud_sincere, interpolation='bilinear')\nax[0].set_title('WordCloud - Câu hỏi chân thành', fontsize=15)\nax[0].axis('off')\n\n# WordCloud cho câu hỏi không chân thành\ninsincere_text = ' '.join(insincere_words)\nwordcloud_insincere = WordCloud(width=800, height=400, background_color='black',\n                              max_words=200, contour_width=3, contour_color='firebrick', colormap='Reds')\nwordcloud_insincere.generate(insincere_text)\nax[1].imshow(wordcloud_insincere, interpolation='bilinear')\nax[1].set_title('WordCloud - Câu hỏi không chân thành', fontsize=15)\nax[1].axis('off')\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-10T01:55:56.881227Z","iopub.execute_input":"2025-05-10T01:55:56.882224Z","iopub.status.idle":"2025-05-10T01:57:08.415413Z","shell.execute_reply.started":"2025-05-10T01:55:56.882184Z","shell.execute_reply":"2025-05-10T01:57:08.414048Z"}},"outputs":[],"execution_count":null}]}