{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Báo cáo bài tập lớn môn Học Máy\n\nHọ và tên: Phú Minh Nhật\n\nMã sinh viên: 18020976\n\nLớp môn học: INT3405_1","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\n%matplotlib inline\nimport matplotlib as mp\nimport matplotlib.pyplot as plt\nimport seaborn as sns","metadata":{"execution":{"iopub.status.busy":"2022-01-08T01:39:09.036940Z","iopub.execute_input":"2022-01-08T01:39:09.037440Z","iopub.status.idle":"2022-01-08T01:39:09.853923Z","shell.execute_reply.started":"2022-01-08T01:39:09.037336Z","shell.execute_reply":"2022-01-08T01:39:09.853221Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Giới thiệu\n**Mô tả vấn đề:**\n\nMột trong những vấn đề hiện hữu trên các website ngày nay là làm cách nào để xử lý những nội dung độc hại và gây chia rẽ. Quora muốn đối mặt và giải quyết vấn đề này để giữ cho nền tảng của họ là một nơi an toàn để người dùng có thể chia sẻ kiến thức ra toàn cầu.\n\nQuora là một nền tảng khuyến khích mọi người học hỏi lẫn nhau. Trên đó, mọi người có thể đặt câu hỏi và kết nối đến những người có thể đóng góp những câu trả lời chất lượng. Thách thức mấu chốt là loại bỏ các câu hỏi không thành thật - của những người hỏi dựa trên những thông tin sai lệch, hoặc nhằm đưa ra ý kiến cá nhân hơn là tìm kiếm câu trả lời.\n\n","metadata":{"id":"CJbfltBuuMgH"}},{"cell_type":"markdown","source":"**Mô tả dữ liệu:**\n\nDữ liệu để train bao gồm rất nhiều câu hỏi đã được hỏi và đã được đánh nhãn là insincere (target = 1) hoặc không (target = 0).\n\n","metadata":{"id":"2Xoa_Yvm1r2L"}},{"cell_type":"code","source":"#Đọc vào dữ liệu dùng để huấn luyện\ndata_raw = pd.read_csv('../input/quora-insincere-questions-classification/train.csv')\ndata_raw.head()","metadata":{"id":"7pHtLSljhLOq","outputId":"c135cebf-523a-4b03-9fa1-69160e36da74","execution":{"iopub.status.busy":"2022-01-08T01:39:09.855545Z","iopub.execute_input":"2022-01-08T01:39:09.855861Z","iopub.status.idle":"2022-01-08T01:39:14.173303Z","shell.execute_reply.started":"2022-01-08T01:39:09.855824Z","shell.execute_reply":"2022-01-08T01:39:14.172618Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"insincere_questions = data_raw[data_raw['target'] == 1].question_text\nsincere_questions = data_raw[data_raw['target'] == 0].question_text","metadata":{"id":"tjMmlWRalN9c","execution":{"iopub.status.busy":"2022-01-08T01:39:14.177318Z","iopub.execute_input":"2022-01-08T01:39:14.179237Z","iopub.status.idle":"2022-01-08T01:39:14.299491Z","shell.execute_reply.started":"2022-01-08T01:39:14.179200Z","shell.execute_reply":"2022-01-08T01:39:14.298698Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Ví dụ về câu hỏi insincere\ninsincere_questions.sample(3, random_state=1).values","metadata":{"id":"On_ZcCUclYfJ","outputId":"006f06dd-5bcd-4482-ec69-05087cc75ed9","execution":{"iopub.status.busy":"2022-01-08T01:39:14.304893Z","iopub.execute_input":"2022-01-08T01:39:14.306862Z","iopub.status.idle":"2022-01-08T01:39:14.319221Z","shell.execute_reply.started":"2022-01-08T01:39:14.306815Z","shell.execute_reply":"2022-01-08T01:39:14.318602Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Ví dụ về câu hỏi sincere\nsincere_questions.sample(3, random_state=1).values","metadata":{"id":"SHf2tvf7la5R","outputId":"a288dd59-688d-480f-90f6-74585c692deb","execution":{"iopub.status.busy":"2022-01-08T01:39:14.323132Z","iopub.execute_input":"2022-01-08T01:39:14.325256Z","iopub.status.idle":"2022-01-08T01:39:14.378398Z","shell.execute_reply.started":"2022-01-08T01:39:14.325196Z","shell.execute_reply":"2022-01-08T01:39:14.377825Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Phân tích và trực quan hóa dữ liệu\n\n**Phân tích dữ liệu thô**\n\nCó thể nhận thấy dữ liệu hiện tại đang thiếu cân bằng, khi mà hầu hết các câu hỏi đều được đánh nhãn sincere, và chỉ có một số lượng nhỏ là insincere","metadata":{"id":"u2bTt7toiuPQ"}},{"cell_type":"code","source":"# Đếm số lượng câu hỏi được đánh nhãn sincere và insincere\nvalues = data_raw.target.value_counts()\nprint(values)\n\n# Tính tỉ lệ giữa số lượng câu hỏi đánh nhãn sincere với câu hỏi đánh nhãn insincere\nsincere_q_pc = values[0]/values.sum()*100\ninsincere_q_pc = values[1]/values.sum()*100\nprint('\\n{}% of questions are sincere while {}% are insincere'.format(sincere_q_pc, insincere_q_pc))","metadata":{"id":"Ka8cmNcClwUq","outputId":"1e914acc-9ca0-40f4-ae5b-98c57f9f89de","execution":{"iopub.status.busy":"2022-01-08T01:39:14.379172Z","iopub.execute_input":"2022-01-08T01:39:14.379389Z","iopub.status.idle":"2022-01-08T01:39:14.406917Z","shell.execute_reply.started":"2022-01-08T01:39:14.379360Z","shell.execute_reply":"2022-01-08T01:39:14.406120Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Vẽ đồ thị miêu tả sự chênh lệch giữa số câu hỏi sincere và insincere\nnames = ['Sincere', 'Insincere']\n\nplt.bar(names, values)\nplt.suptitle('Number of Sincere and Insincere Questions')\nplt.show()\n","metadata":{"id":"LDkux6t2l4TB","outputId":"ea382e1c-9e6a-486d-dbfb-5ea757db0e6e","execution":{"iopub.status.busy":"2022-01-08T01:39:14.409580Z","iopub.execute_input":"2022-01-08T01:39:14.409780Z","iopub.status.idle":"2022-01-08T01:39:14.578858Z","shell.execute_reply.started":"2022-01-08T01:39:14.409757Z","shell.execute_reply":"2022-01-08T01:39:14.578180Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Import thư viện wordcloud\nfrom wordcloud import WordCloud, ImageColorGenerator\n\n# Tách các câu hỏi thành các dictionary chứa các từ độc nhất và tần suất xuất hiện\ndef word_freq_dict(text):\n    # Convert text into word list\n    wordList = text.split()\n    # Generate word freq dictionary\n    wordFreqDict = {word: wordList.count(word) for word in wordList}\n    return wordFreqDict","metadata":{"id":"OavvnE8smFgI","execution":{"iopub.status.busy":"2022-01-08T01:39:14.580074Z","iopub.execute_input":"2022-01-08T01:39:14.580307Z","iopub.status.idle":"2022-01-08T01:39:14.627031Z","shell.execute_reply.started":"2022-01-08T01:39:14.580275Z","shell.execute_reply":"2022-01-08T01:39:14.626395Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Vẽ wordcloud của dictionary chứa tần suất của từ\ndef word_cloud_from_frequency(word_freq_dict, title, figure_size=(10,6)):\n    wordcloud.generate_from_frequencies(word_freq_dict)\n    plt.figure(figsize=figure_size)\n    plt.imshow(wordcloud)\n    plt.axis(\"off\")\n    plt.title(title)\n    plt.show()","metadata":{"id":"SBvGniUbnCi5","execution":{"iopub.status.busy":"2022-01-08T01:39:14.628284Z","iopub.execute_input":"2022-01-08T01:39:14.628543Z","iopub.status.idle":"2022-01-08T01:39:14.633994Z","shell.execute_reply.started":"2022-01-08T01:39:14.628509Z","shell.execute_reply":"2022-01-08T01:39:14.633180Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Wordcloud của 1 tập 1000 câu hỏi insincere ngẫu nhiên\ninsincere_questions = data_raw.question_text[data_raw['target'] == 1]\ninsincere_sample = \" \".join(insincere_questions.sample(1000, random_state=1).values)\ninsincere_word_freq = word_freq_dict(insincere_sample)\nwordcloud = WordCloud(width= 5000,\n    height=3000,\n    max_words=200,\n    colormap='Reds',\n    background_color='white')\nword_cloud_from_frequency(insincere_word_freq, \"Các từ xuất hiện nhiều nhất trong tập 1000 câu hỏi đánh nhãn insincere\") ","metadata":{"id":"6zLLY3VjnLvb","outputId":"677f132b-1cff-4592-a951-9ff7139b85d9","execution":{"iopub.status.busy":"2022-01-08T01:39:14.637608Z","iopub.execute_input":"2022-01-08T01:39:14.638105Z","iopub.status.idle":"2022-01-08T01:40:01.366685Z","shell.execute_reply.started":"2022-01-08T01:39:14.638070Z","shell.execute_reply":"2022-01-08T01:40:01.366008Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Wordcloud của tập 1000 câu hỏi sincere ngẫu nhiên\nsincere_questions = data_raw.question_text[data_raw['target'] == 0]\nsincere_sample = \" \".join(sincere_questions.sample(1000, random_state=1).values)\nsincere_word_freq = word_freq_dict(sincere_sample)\nwordcloud = WordCloud(width= 5000,\n    height=3000,\n    max_words=200,\n    colormap='Greens',\n    background_color='white')\n\nword_cloud_from_frequency(sincere_word_freq, \"Các từ xuất hiện nhiều nhất trong tập 1000 câu hỏi chưa xử lý đánh nhãn sincere\")","metadata":{"id":"1dzmy2GgnjU5","outputId":"e1607bd2-531c-46dd-bca0-44983cd3604b","execution":{"iopub.status.busy":"2022-01-08T01:40:01.367701Z","iopub.execute_input":"2022-01-08T01:40:01.367941Z","iopub.status.idle":"2022-01-08T01:40:42.001366Z","shell.execute_reply.started":"2022-01-08T01:40:01.367912Z","shell.execute_reply":"2022-01-08T01:40:42.000469Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Có thể thấy, những từ xuất hiện nhiều nhất là những từ được sử dụng rất nhiều và thường xuyên như 'what', 'is', 'with', 'are',... nên không có tác dụng gì trong mô hình. Vậy nên, những từ phổ biến này (còn gọi là stopword) cần phải được loại bỏ.","metadata":{"id":"dFtFkO67iD0C"}},{"cell_type":"code","source":"import nltk\nimport sys\nimport spacy\n\nnltk.download('stopwords')\nnltk.download('averaged_perceptron_tagger')\nnltk.download('wordnet')\n\nfrom nltk.corpus import stopwords\nfrom nltk.stem.porter import PorterStemmer\nimport string","metadata":{"id":"urIjnjoRn5Mz","outputId":"ea5dd1d1-9620-4946-cbcd-73374f4cf197","execution":{"iopub.status.busy":"2022-01-08T01:40:42.002756Z","iopub.execute_input":"2022-01-08T01:40:42.003144Z","iopub.status.idle":"2022-01-08T01:41:51.798187Z","shell.execute_reply.started":"2022-01-08T01:40:42.003107Z","shell.execute_reply":"2022-01-08T01:41:51.797430Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Để chuẩn hóa văn bản và giảm thiểu độ nhiễu, quá trình tiền xử lý được áp dụng theo các bước sau:\n\n*   Bước 1: Chuyển tất cả các ký tự về ký tự in thường\n*   Bước 2: Chia các văn bản thành các list\n*   Bước 3: Loại bỏ toàn bộ các dấu câu\n*   Bước 4: Xóa tất cả các stopword có trong câu\n*   Bước 5: Stemming - một kỹ thuật trích xuất từ cơ sở bằng cách loại bỏ các phụ tố của từ\n\nCông cụ được sử dụng để thực hiện tiền xử lý là NTLK (Natural Language Toolkit)\n\n\n","metadata":{"id":"Mkqs94geleWo"}},{"cell_type":"code","source":"nlp = spacy.load(\"en_core_web_sm\", disable=['parser','ner'])\nstop = set(stopwords.words('english'))\npunc = set(string.punctuation)\n\ndef clean_text(text):\n    # Chuyển toàn bộ văn bản sang chữ in thường\n    text = text.lower()\n    # Tách câu văn thành list các từ\n    wordList = text.split()\n    # Loại bỏ các dấu câu\n    wordList = [\"\".join(x for x in word if (x==\"'\")|(x not in punc)) for word in wordList]\n    # Loại bỏ stop words\n    wordList = [word for word in wordList if word not in stop]\n    # Stem\n    porter = PorterStemmer()\n    wordList = [porter.stem(word) for word in wordList]\n\n    reformed_sentence = \" \".join(wordList)\n    doc = nlp(reformed_sentence)\n    return \" \".join([token.lemma_ for token in doc])","metadata":{"id":"g6i39rsZoCxp","execution":{"iopub.status.busy":"2022-01-08T01:41:51.799357Z","iopub.execute_input":"2022-01-08T01:41:51.799878Z","iopub.status.idle":"2022-01-08T01:41:52.519218Z","shell.execute_reply.started":"2022-01-08T01:41:51.799847Z","shell.execute_reply":"2022-01-08T01:41:52.518507Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"question = data_raw.question_text.sample(1, random_state=1).values[0]\nquestion","metadata":{"id":"ALheCR05oMFw","outputId":"4c0278d6-2063-4aa2-ea9c-58723e8469ad","execution":{"iopub.status.busy":"2022-01-08T01:41:52.520401Z","iopub.execute_input":"2022-01-08T01:41:52.522397Z","iopub.status.idle":"2022-01-08T01:41:52.556377Z","shell.execute_reply.started":"2022-01-08T01:41:52.522360Z","shell.execute_reply":"2022-01-08T01:41:52.555609Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Kiểm tra xem hàm có hoạt động hay không\nclean_text(question)","metadata":{"id":"vQUFOn3ioaSf","outputId":"ca57a6b8-79e6-46f5-8284-1e5621edbd52","execution":{"iopub.status.busy":"2022-01-08T01:41:52.557677Z","iopub.execute_input":"2022-01-08T01:41:52.557993Z","iopub.status.idle":"2022-01-08T01:41:52.578750Z","shell.execute_reply.started":"2022-01-08T01:41:52.557956Z","shell.execute_reply":"2022-01-08T01:41:52.578075Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Thực hiện clean toàn bộ dữ liệu\ndata_raw['clean_text'] = data_raw['question_text'].astype('str').apply(clean_text)","metadata":{"id":"dgttouKOoqkr","execution":{"iopub.status.busy":"2022-01-08T01:41:52.579940Z","iopub.execute_input":"2022-01-08T01:41:52.580723Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_raw.clean_text.head()","metadata":{"id":"g_ntyKL8Sgw-","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Wordcloud của tập 1000 câu hỏi insincere ngẫu nhiên đã clean\nclean_insincere_questions = data_raw.clean_text[data_raw['target'] == 1]\nclean_insincere_sample = \" \".join(clean_insincere_questions.sample(1000, random_state=1).values)\nclean_insincere_word_freq = word_freq_dict(clean_insincere_sample)\nwordcloud = WordCloud(width= 5000,\n    height=3000,\n    max_words=200,\n    colormap='Reds',\n    background_color='white')\n\nword_cloud_from_frequency(clean_insincere_word_freq, \"Các từ xuất hiện nhiều nhất trong tập 1000 câu hỏi đã làm sạch đánh nhãn insincere\") ","metadata":{"id":"vwIGp1g9SlEx","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Wordcloud của tập 1000 câu hỏi sincere ngẫu nhiên đã clean\nclean_sincere_questions = data_raw.clean_text[data_raw['target'] == 0]\nclean_sincere_sample = \" \".join(clean_sincere_questions.sample(1000, random_state=1).values)\nclean_sincere_word_freq = word_freq_dict(clean_sincere_sample)\nwordcloud = WordCloud(width= 5000,\n    height=3000,\n    max_words=200,\n    colormap='Greens',\n    background_color='white')\n\nword_cloud_from_frequency(clean_sincere_word_freq, \"Các từ xuất hiện nhiều nhất trong tập 1000 câu hỏi đã làm sạch đánh nhãn sincere\") ","metadata":{"id":"WBIn-lk2S3Fi","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Text2Vec\nCần phải biến đổi dữ liệu từ dạng text sang dạng vector\n\n**Bag Of Word**\n\nBOW là một mô hình cơ bản được dùng trong tác vụ xử lý ngôn ngữ tự nhiên. Nó được gọi như vậy vì mô hình này bỏ đi thứ tự của từ trong văn bản, chỉ thể hiện sự xuất hiện và tần suất xuất hiện của từ trong văn bản.","metadata":{"id":"GzThlCdIvZ7w"}},{"cell_type":"code","source":"from sklearn.feature_extraction.text import CountVectorizer\nbow_converter = CountVectorizer()","metadata":{"id":"7RBEmo1lTPmk","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_question_text = data_raw['clean_text'].sample(1, random_state= 1).values\nsample_question_text","metadata":{"id":"ouQfYDZkTRPk","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_count_vectorized_data = bow_converter.fit_transform(sample_question_text)\nsample_count_vectorized_data.toarray()","metadata":{"id":"FtvsRr2uTW8r","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"count_vectorized_data_feature_names = bow_converter.get_feature_names()\ncount_vectorized_data_feature_names","metadata":{"id":"OfIHFVloTeys","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# TF-IDF\nTF-IDF (viết tắt của term frequency – inverse document frequency) là một phương thức thống kê thường được sử dụng trong mảng truy xuất thông tin (information retrieval) và khai phá dữ liệu văn bản (text mining) để đánh giá mức độ quan trọng của một cụm từ đối với một tài liệu cụ thể trong một tập hợp bao gồm nhiều tài liệu. Phương thức này bao gồm 2 khái niệm:\n\n+ TF (Term Frequency - Tần suất xuất hiện của từ) là số lần từ xuất hiện trong văn bản. Vì các văn bản có thể có độ dài ngắn khác nhau nên một số từ có thể xuất hiện nhiều lần trong một văn bản dài hơn là một văn bản ngắn. Như vậy, term frequency thường được chia cho độ dài văn bản( tổng số từ trong một văn bản).\n\n+ IDF (Inverse Document Frequency - Nghịch đảo tần suất của văn bản), giúp đánh giá tầm quan trọng của một từ . Khi tính toán TF , tất cả các từ được coi như có độ quan trọng bằng nhau. Nhưng  một số từ như “is”, “of” và “that” thường xuất hiện rất nhiều lần nhưng độ quan trọng là không cao. Như thế chúng ta cần giảm độ quan trọng của những từ này xuống.","metadata":{"id":"4iwu78IOsFxX"}},{"cell_type":"code","source":"from sklearn.feature_extraction.text import TfidfVectorizer\ntfidf_converter = TfidfVectorizer(ngram_range=(1,1))","metadata":{"id":"LXRcmr9-TqFO","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_tfidf_vectorized_data = tfidf_converter.fit_transform(sample_question_text)\nsample_tfidf_vectorized_data.toarray()","metadata":{"id":"Fw8lyRsbT0aL","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tfidf_word_feature_names = tfidf_converter.get_feature_names()\ntfidf_word_feature_names","metadata":{"id":"5N1UgjCmT6HK","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(tfidf_word_feature_names)","metadata":{"id":"xmmGCobbT9gS","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Mô hình\nMô hình trong bài toán này sẽ sử dụng thuật toán Hồi quy Logistics (Logistics Regression). Tìm hiểu sâu hơn về LR:\n\nhttps://excessive-source-1c9.notion.site/16-09-2021-H-i-quy-Logistics-cdcc911147e5458ba9203b58e6bd0099\n","metadata":{"id":"37RiWSZ8tKSv"}},{"cell_type":"code","source":"from sklearn.linear_model import LogisticRegression\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.metrics import accuracy_score, confusion_matrix\nfrom sklearn.model_selection import train_test_split\n\ncount_vectorizer = CountVectorizer()\nmodel = LogisticRegression(C=1, random_state=0, max_iter=1000)\n\nvectorize_logit_pipeline = Pipeline([\n    ('count_vectorizer', count_vectorizer),\n    ('logit', model)\n])","metadata":{"id":"c20AiZZ0UDSK","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Pipeline với LR và Count Vectorizer\n","metadata":{"id":"yD6VmQI5uoX5"}},{"cell_type":"code","source":"# Biến đầu vào\nX = data_raw['clean_text']\n# Biến đầu ra\ny = data_raw['target']","metadata":{"id":"4x3at6rGYNqL","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Chia bộ dữ liệu huấn luyện thành 2 tập, với 1 tập để huẩn luyện, và 1 tập để kiểm thử","metadata":{"id":"PzIgLQk4vjt3"}},{"cell_type":"code","source":"train_X, test_X, train_y, test_y = train_test_split(X, y, test_size=0.3)","metadata":{"id":"dY8wVkN9YQun","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Huấn luyện mô hình sử dụng tập dữ liệu huấn luyện và đặc trưng","metadata":{"id":"JjFVlFw0vkDn"}},{"cell_type":"code","source":"vectorize_logit_pipeline.fit(train_X, train_y)","metadata":{"id":"0dAApJ5SYS51","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Đưa ra dự đoán của mô hình","metadata":{"id":"xwc1Cjdbv6C4"}},{"cell_type":"code","source":"predictions = vectorize_logit_pipeline.predict(test_X)","metadata":{"id":"TC3dCyxDZLNd","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Kiểm tra điểm chính xác","metadata":{"id":"FlXzfaJEv-o3"}},{"cell_type":"code","source":"accuracy_score(test_y, predictions)","metadata":{"id":"VbD56N58ZNJb","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Kiểm tra điểm F1","metadata":{"id":"yEkKlfjywCUn"}},{"cell_type":"code","source":"from sklearn.metrics import f1_score\nf1_score(test_y, predictions)","metadata":{"id":"9EyXlvisZPas","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Vẽ ma trận confusion","metadata":{"id":"nrRo1HMmwEyO"}},{"cell_type":"code","source":"confusion_matrix_logit_tfidf = confusion_matrix(test_y, predictions)\nsns.heatmap(confusion_matrix_logit_tfidf, annot= True, xticklabels=['sincere', 'insincere'], yticklabels=['sincere', 'insincere'])","metadata":{"id":"0Bheio7aZhAT","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import classification_report\nprint(classification_report(test_y, predictions))","metadata":{"id":"T1bWsp9lbveZ","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Pipeline sử dụng LR và TF-IDF Bi-gram Vectorizer","metadata":{"id":"mxZoYcqUzKoo"}},{"cell_type":"code","source":"tfidf_ngrams_converter = TfidfVectorizer(ngram_range=(1,2))\ntfidf_ngrams_logit_pipeline = Pipeline([\n    ('tfidf_vectorizer', tfidf_ngrams_converter),\n    ('logit', model)\n])","metadata":{"id":"oelQfm5EgvNJ","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tfidf_ngrams_logit_pipeline.fit(train_X, train_y)","metadata":{"id":"PJs0vRwKg1oN","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"new_predictions = tfidf_ngrams_logit_pipeline.predict(test_X)","metadata":{"id":"i7eU6MIRhiCD","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"accuracy_score(test_y, new_predictions)","metadata":{"id":"MRiAA0Edhmfj","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"f1_score(test_y, new_predictions)","metadata":{"id":"YYuM19tRhnBV","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"confusion_matrix_logit_tfidf = confusion_matrix(test_y, new_predictions)\nsns.heatmap(confusion_matrix_logit_tfidf, annot= True, xticklabels=['sincere', 'insincere'], yticklabels=['sincere', 'insincere'])","metadata":{"id":"0fHojUgwht7L","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(classification_report(test_y, new_predictions, target_names=['sincere', 'insincere']))","metadata":{"id":"ewVRRwCNh2XC","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Có thể thấy, mô hình khi áp dụng các vector đặc trưng Bi-gram cho điểm chính xác và điểm F1 cao hơn","metadata":{"id":"XWVcU53l5M2f"}},{"cell_type":"code","source":"test_data = pd.read_csv('../input/quora-insincere-questions-classification/test.csv')\ntest_data.head()","metadata":{"id":"fT1Z-QdBh4s1","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data.info()","metadata":{"id":"2iH2DSGriE4B","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data['clean_text'] = test_data['question_text'].astype('str').apply(clean_text)","metadata":{"id":"rIfjkRp6iMca","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data.head()","metadata":{"id":"SPg8uXiDnoxu","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_final = test_data['clean_text']","metadata":{"id":"Z0EsW8YtnvAe","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_final = tfidf_ngrams_logit_pipeline.predict(x_final)","metadata":{"id":"E7rxq0kFn_7z","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data['target'] = y_final","metadata":{"id":"Q-4UMWUXogpU","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"result_df = test_data[['qid','target']]","metadata":{"id":"MmLXHoPIov6Q","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"result_df.rename(columns={'target': 'prediction'}, inplace=True)\nresult_df.set_index('qid', inplace=True)\nresult_df.head()","metadata":{"id":"w0GhfjINo5Vf","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"result_df.to_csv('submission.csv')\n!head submission.csv","metadata":{"id":"NmwbwOV63OE7","trusted":true},"execution_count":null,"outputs":[]}]}