{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"execution":{"iopub.status.busy":"2021-12-21T13:24:58.094225Z","iopub.execute_input":"2021-12-21T13:24:58.095004Z","iopub.status.idle":"2021-12-21T13:24:58.105539Z","shell.execute_reply.started":"2021-12-21T13:24:58.094958Z","shell.execute_reply":"2021-12-21T13:24:58.104643Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Mô tả bài toán\n\n\nNhiệm vụ phải làm trong bài toán là phân loại các câu insincere và sincere có trên hệ thống của Quora","metadata":{}},{"cell_type":"markdown","source":"# 1. Phân tích dữ liệu","metadata":{}},{"cell_type":"code","source":"train_data = pd.read_csv(\"../input/quora-insincere-questions-classification/train.csv\")\ntest_data = pd.read_csv(\"../input/quora-insincere-questions-classification/test.csv\")","metadata":{"execution":{"iopub.status.busy":"2021-12-21T13:24:58.108600Z","iopub.execute_input":"2021-12-21T13:24:58.109166Z","iopub.status.idle":"2021-12-21T13:25:01.259268Z","shell.execute_reply.started":"2021-12-21T13:24:58.109129Z","shell.execute_reply":"2021-12-21T13:25:01.258541Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* Như đã nêu ở trên nhiệm vụ phải làm trong bài toán là phân loại các câu insincere và sincere có trên hệ thống của Quora.\n* **Input** là tập các câu hỏi tiếng anh được cho dưới dạng text và đi kèm là **id** của từng câu cũng như nhãn **label** của từng câu. \n* **output** là các giá trị Label **Sincere** (0), **Insincere** (1).","metadata":{}},{"cell_type":"markdown","source":"##  1.1 Tập Train\n* Dữ liệu có thuộc tính qid, question_text, target\n* Khi phân loại, ta dùng thuộc tính question_text là đầu vào X, target là label y","metadata":{}},{"cell_type":"code","source":"train_data","metadata":{"execution":{"iopub.status.busy":"2021-12-21T13:25:01.260652Z","iopub.execute_input":"2021-12-21T13:25:01.260986Z","iopub.status.idle":"2021-12-21T13:25:01.273632Z","shell.execute_reply.started":"2021-12-21T13:25:01.260950Z","shell.execute_reply":"2021-12-21T13:25:01.272901Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Loại bỏ các thành phần dữ liệu NA trong data\ntrain_data.dropna(inplace = True)\ntrain_data","metadata":{"execution":{"iopub.status.busy":"2021-12-21T13:25:01.275179Z","iopub.execute_input":"2021-12-21T13:25:01.275696Z","iopub.status.idle":"2021-12-21T13:25:01.640400Z","shell.execute_reply.started":"2021-12-21T13:25:01.275660Z","shell.execute_reply":"2021-12-21T13:25:01.639722Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Sau khi loại bỏ các dữ liệu NA chúng ta sẽ cùng xem các câu có nhãn là 1 và 0","metadata":{}},{"cell_type":"code","source":"# Lấy ra câu có nhãn 1(Insincere)\ntoxic_data = train_data[train_data.target == 1]\n## Lấy ra câu có nhãn 0(Sincere)\nnon_toxic_data = train_data[train_data.target == 0]","metadata":{"execution":{"iopub.status.busy":"2021-12-21T13:25:01.642748Z","iopub.execute_input":"2021-12-21T13:25:01.643168Z","iopub.status.idle":"2021-12-21T13:25:01.736371Z","shell.execute_reply.started":"2021-12-21T13:25:01.643130Z","shell.execute_reply":"2021-12-21T13:25:01.735652Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Insincere** **Label target = 1:** ","metadata":{}},{"cell_type":"code","source":"for sentence in toxic_data.question_text.sample(5):\n  print(sentence)","metadata":{"execution":{"iopub.status.busy":"2021-12-21T13:25:01.737827Z","iopub.execute_input":"2021-12-21T13:25:01.738087Z","iopub.status.idle":"2021-12-21T13:25:01.747200Z","shell.execute_reply.started":"2021-12-21T13:25:01.738053Z","shell.execute_reply":"2021-12-21T13:25:01.746474Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Sincere** **Label target = 0:** ","metadata":{}},{"cell_type":"code","source":"for sentence in non_toxic_data.question_text.sample(5):\n  print(sentence)","metadata":{"execution":{"iopub.status.busy":"2021-12-21T13:25:01.748575Z","iopub.execute_input":"2021-12-21T13:25:01.749148Z","iopub.status.idle":"2021-12-21T13:25:01.788253Z","shell.execute_reply.started":"2021-12-21T13:25:01.749114Z","shell.execute_reply":"2021-12-21T13:25:01.787555Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Tỷ lệ của các câu Sincere và Insincere**","metadata":{}},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\n# Mô phỏng về độ tương quan giữa các câu insincere và sincere dưới dạng biểu đồ:\n\nval = train_data.target.value_counts().values\nnames = ['Sincere', 'Insincere']\nplt.bar(names, val)\nplt.suptitle('Number of Sincere and Insincere Questions')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-12-21T13:25:01.789860Z","iopub.execute_input":"2021-12-21T13:25:01.790134Z","iopub.status.idle":"2021-12-21T13:25:01.943769Z","shell.execute_reply.started":"2021-12-21T13:25:01.790100Z","shell.execute_reply":"2021-12-21T13:25:01.942950Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Phân bố của dữ liệu","metadata":{}},{"cell_type":"code","source":"labels = 'Sincere', 'Insincere'\nsizes = [(non_toxic_data.shape[0] / train_data.shape[0])*100, (toxic_data.shape[0] / train_data.shape[0])*100]\nplt.pie(sizes, labels=labels,\nautopct='%1.1f%%', shadow=True, startangle=140)\nplt.axis('equal')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-12-21T13:25:01.948344Z","iopub.execute_input":"2021-12-21T13:25:01.948685Z","iopub.status.idle":"2021-12-21T13:25:02.056813Z","shell.execute_reply.started":"2021-12-21T13:25:01.948643Z","shell.execute_reply":"2021-12-21T13:25:02.056148Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Nhận xét\nTrong dữ liệu train, có tới **93,8%** dữ liệu **sincere** (label 0) nhưng chỉ có **6,2%** dữ liệu **Insincere** (label 1).\n\n**Kết luận:** Mất cân bằng về dữ liệu => Độ đo sử dụng là F1-Score.\n**F1_score** là trung bình điều hòa giữa precision (độ chính xác) và recall (độ bao phủ)\n\n**Precision**: trong tập tìm được thì bao nhiêu cái (phân loại) đúng.\n\n**Recall:** trong số các tồn tại, tìm ra được bao nhiêu cái (phân loại).","metadata":{"execution":{"iopub.status.busy":"2021-12-20T13:47:52.492929Z","iopub.execute_input":"2021-12-20T13:47:52.49326Z","iopub.status.idle":"2021-12-20T13:47:52.499777Z","shell.execute_reply.started":"2021-12-20T13:47:52.493222Z","shell.execute_reply":"2021-12-20T13:47:52.498497Z"}}},{"cell_type":"markdown","source":"# 2. Mô tả thuật toán\n## Tiền xử lý dữ liệu \n* Unicode: Chuyển về dạng unicode\n* Lowercase : Chuyển về dạng in thường.\n* Punctuation, Remove Number: Bỏ dấu, chữ số.\n* Tokenize, Stopwords: Tách từ, Các từ dừng.\n* Lemmatizers: Rút gọn từ về dạng ngắn gọn.\n* => Loại bỏ các từ không ý nghĩa, giúp ta xác định được các từ cần thiết để phân loại từ đó thuộc Label 1 hay 0.\n\n# Vector hóa dữ liệu \n* Sử dụng CountVectorizer để trích xuất các từ, biến words thành dạng vectors ở dạng Bag-of-Words bằng cách đếm số lần xuất hiện của các từ trong bộ dữ liệu.\n\n* TF-IDF (Term Frequency – Inverse Document Frequency) là 1 kĩ thuật sử dụng trong khai phá dữ liệu văn bản. Trọng số này được sử dụng để đánh giá tầm quan trọng của một từ trong một văn bản. Giá trị cao thể hiện độ quan trọng cao và nó phụ thuộc vào số lần từ xuất hiện trong văn bản nhưng bù lại bởi tần suất của từ đó trong tập dữ liệu.\n \n#  Huấn luyện mô hình\n* Áp dụng mô hình học máy Logistic Regression.","metadata":{}},{"cell_type":"markdown","source":"### 2.1. Tiền xử lý\n* Sử dụng thư viện nltk(Bộ công cụ ngôn ngữ tự nhiên).","metadata":{}},{"cell_type":"code","source":"import re\nimport nltk\nimport string\nfrom unidecode import unidecode\nfrom nltk.tokenize import word_tokenize\nfrom nltk.corpus import stopwords\nfrom nltk.stem import PorterStemmer, WordNetLemmatizer\n\ncontraction_dict = {\"dont\": \"do not\", \"aint\": \"is not\", \"isnt\": \"is not\", \"doesnt\": \"does not\"\n, \"cant\": \"cannot\", \"mustnt\": \"must not\", \"ll\":\"will\" , \"re\": \"are\" ,\"ll\": \"will\", \"wont\": \"will not\" ,\"hasnt\": \"has not\"\n, \"havent\": \"have not\", \"arent\": \"are not\", \"ain't\": \"is not\", \"aren't\": \"are not\"\n,\"can't\": \"cannot\", \"‘cause\": \"because\", \"could've\": \"could have\"\n, \"couldn't\": \"could not\", \"didn't\": \"did not\", \"doesn't\": \"does not\", \"don't\": \"do not\"\n, \"hadn't\": \"had not\", \"hasn't\": \"has not\", \"haven't\": \"have not\", \"he'd\": \"he would\"\n,\"he'll\": \"he will\", \"he's\": \"he is\", \"how'd\": \"how did\", \"how'd'y\": \"how do you\"\n, \"how'll\": \"how will\", \"how's\": \"how is\", \"I'd\": \"I would\", \"I'd've\": \"I would have\"\n, \"I'll\": \"I will\", \"I'll've\": \"I will have\",\"I'm\": \"I am\", \"Iam\": \"I am\", \"I've\": \"I have\"\n, \"i'd\": \"i would\", \"i'd've\": \"i would have\", \"i'll\": \"i will\", \"i'll've\": \"i will have\"\n,\"i'm\": \"i am\", \"i've\": \"i have\", \"isn't\": \"is not\", \"it'd\": \"it would\", \"it'd've\": \"it would have\"\n, \"it'll\": \"it will\", \"it'll've\": \"it will have\",\"it's\": \"it is\", \"let's\": \"let us\"\n, \"ma'am\": \"madam\", \"mayn't\": \"may not\", \"might've\": \"might have\"\n,\"mightn't\": \"might not\",\"mightn't've\": \"might not have\", \"must've\": \"must have\"\n, \"mustn't\": \"must not\", \"mustn't've\": \"must not have\", \"needn't\": \"need not\"\n, \"needn't've\": \"need not have\",\"o'clock\": \"of the clock\", \"oughtn't\": \"ought not\"\n, \"oughtn't've\": \"ought not have\", \"shan't\": \"shall not\", \"sha'n't\": \"shall not\"\n, \"shan't've\": \"shall not have\", \"she'd\": \"she would\", \"she'd've\": \"she would have\"\n, \"she'll\": \"she will\", \"she'll've\": \"she will have\", \"she's\": \"she is\"\n, \"should've\": \"should have\", \"shouldn't\": \"should not\", \"shouldn't've\": \"should not have\"\n, \"so've\": \"so have\",\"so's\": \"so as\", \"this's\": \"this is\",\"that'd\": \"that would\", \n\"that'd've\": \"that would have\", \"that's\": \"that is\", \"there'd\": \"there would\", \n\"there'd've\": \"there would have\", \"there's\": \"there is\", \"here's\": \"here is\",\"they'd\": \"they would\"\n, \"they'd've\": \"they would have\", \"they'll\": \"they will\", \"they'll've\": \"they will have\", \n\"they're\": \"they are\", \"they've\": \"they have\", \"to've\": \"to have\", \"wasn't\": \"was not\",\n\"we'd\": \"we would\", \"we'd've\": \"we would have\", \"we'll\": \"we will\", \n\"we'll've\": \"we will have\", \"we're\": \"we are\", \"we've\": \"we have\", \n\"weren't\": \"were not\", \"what'll\": \"what will\", \"what'll've\": \"what will have\", \n\"what're\": \"what are\", \"what's\": \"what is\", \"what've\": \"what have\", \"when's\": \"when is\",\n\"when've\": \"when have\", \"where'd\": \"where did\", \"where's\": \"where is\", \"where've\": \"where have\", \n\"who'll\": \"who will\", \"who'll've\": \"who will have\", \"who's\": \"who is\", \"who've\": \"who have\",\n\"why's\": \"why is\", \"why've\": \"why have\", \"will've\": \"will have\", \"won't\": \"will not\", \"won't've\": \"will not have\",\n\"would've\": \"would have\", \"wouldn't\": \"would not\", \"wouldn't've\": \"would not have\", \"y'all\": \"you all\", \n\"y'all'd\": \"you all would\",\"y'all'd've\": \"you all would have\",\"y'all're\": \"you all are\",\n\"y'all've\": \"you all have\", \"you'd\": \"you would\", \"you'd've\": \"you would have\", \n\"you'll\": \"you will\", \"youll\":\"you will\", \"you'll've\": \"you will have\", \"you're\": \"you are\", \"you've\": \"you have\"}\n\n\nnltk.download('stopwords')\nnltk.download('punkt')\nnltk.download('wordnet')\nnltk_stopwords = stopwords.words('english')\n\nnltk_stopwords.remove('not')\n\nstemmer = PorterStemmer()\nlemmatizer = WordNetLemmatizer()","metadata":{"execution":{"iopub.status.busy":"2021-12-21T13:25:02.059795Z","iopub.execute_input":"2021-12-21T13:25:02.060002Z","iopub.status.idle":"2021-12-21T13:25:02.079836Z","shell.execute_reply.started":"2021-12-21T13:25:02.059977Z","shell.execute_reply":"2021-12-21T13:25:02.078637Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Clean dữ liệu","metadata":{}},{"cell_type":"code","source":"def clean(text):        \n    # chuyển từ về dạng unicode \n    text = unidecode(text).encode(\"ascii\")\n    text = str(text, \"ascii\")\n\n    # chuyển về dạng chữ thường, kí tự đặc biệt, chữ số.\n    text = text.lower()\n    text = re.sub('https?://\\S+|www\\.\\S+', '', text)\n    text = re.sub('<.*?>+', '', text)\n    text = re.sub('[%s]' % re.escape(string.punctuation), '', text)  \n    text = re.sub('\\n', '', text)\n    text = re.sub('[’“”…]', ' ', text)  \n    text = ''.join(i for i in text if not i.isdigit())\n\n    # Chuyển các từ viết tắt trong từ điển về dạng thường\n    tokens = word_tokenize(text)\n    tokens = [contraction_dict.get(token) if (contraction_dict.get(token) != None) else token for token in tokens]\n    text = \" \".join(tokens)\n\n    # Bỏ các từ chứa ở trong stop-words   \n    tokens = word_tokenize(text)\n    tokens_without_sw = [word for word in tokens if not word in nltk_stopwords]\n\n    # Chuyển từ về dạng số nhiều về dạng thường\n    text = [lemmatizer.lemmatize(word) for word in tokens_without_sw ] \n    text = \" \".join(text)\n\n    return text","metadata":{"execution":{"iopub.status.busy":"2021-12-21T13:25:02.081207Z","iopub.execute_input":"2021-12-21T13:25:02.081447Z","iopub.status.idle":"2021-12-21T13:25:02.091939Z","shell.execute_reply.started":"2021-12-21T13:25:02.081416Z","shell.execute_reply":"2021-12-21T13:25:02.091208Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Dữ liệu sau khi được clean","metadata":{}},{"cell_type":"code","source":"train_data['clean_questions'] = train_data['question_text'].apply(clean)\ntrain_data","metadata":{"execution":{"iopub.status.busy":"2021-12-21T13:25:02.093474Z","iopub.execute_input":"2021-12-21T13:25:02.093786Z","iopub.status.idle":"2021-12-21T13:33:56.749366Z","shell.execute_reply.started":"2021-12-21T13:25:02.093750Z","shell.execute_reply":"2021-12-21T13:33:56.748722Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = train_data['clean_questions']\ny = train_data['target']\nSincere_data = X[y == 0]\nInsincere_data = X[y == 1]","metadata":{"execution":{"iopub.status.busy":"2021-12-21T13:33:56.752166Z","iopub.execute_input":"2021-12-21T13:33:56.752686Z","iopub.status.idle":"2021-12-21T13:33:56.787861Z","shell.execute_reply.started":"2021-12-21T13:33:56.752657Z","shell.execute_reply":"2021-12-21T13:33:56.787187Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from collections import Counter\n\n# top 20 words in a Sincere data\np = Counter(\" \".join(Sincere_data).split()).most_common(20)\nrslt = pd.DataFrame(p, columns=['Word', 'Frequency'])\n\nrslt.plot(x='Word',kind = \"barh\", figsize=(12,10), title=\"Top 20 từ sử dụng nhiều nhất của label 0\")","metadata":{"execution":{"iopub.status.busy":"2021-12-21T13:33:56.789123Z","iopub.execute_input":"2021-12-21T13:33:56.789434Z","iopub.status.idle":"2021-12-21T13:33:59.334517Z","shell.execute_reply.started":"2021-12-21T13:33:56.789399Z","shell.execute_reply":"2021-12-21T13:33:59.333744Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from collections import Counter\n\n# top 20 words in a Insincere data\np = Counter(\" \".join(Sincere_data).split()).most_common(20)\nrslt = pd.DataFrame(p, columns=['Word', 'Frequency'])\n\nrslt.plot(x='Word',kind = \"barh\", figsize=(12,10), title=\"Top 20 từ sử dụng nhiều nhất của label 1\")","metadata":{"execution":{"iopub.status.busy":"2021-12-21T13:33:59.335967Z","iopub.execute_input":"2021-12-21T13:33:59.336205Z","iopub.status.idle":"2021-12-21T13:34:01.773493Z","shell.execute_reply.started":"2021-12-21T13:33:59.336172Z","shell.execute_reply":"2021-12-21T13:34:01.772737Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":" ## VECTOR HÓA DỮ LIỆU \n","metadata":{}},{"cell_type":"code","source":"from sklearn.feature_extraction.text import CountVectorizer, TfidfVectorizer\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.metrics import accuracy_score, f1_score,recall_score, confusion_matrix, classification_report, plot_confusion_matrix\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import classification_report","metadata":{"execution":{"iopub.status.busy":"2021-12-21T13:59:01.421470Z","iopub.execute_input":"2021-12-21T13:59:01.421803Z","iopub.status.idle":"2021-12-21T13:59:01.427421Z","shell.execute_reply.started":"2021-12-21T13:59:01.421766Z","shell.execute_reply":"2021-12-21T13:59:01.426548Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Sử dụng thêm cách vector hóa dữ liệu bằng phương pháp TF_IDF","metadata":{}},{"cell_type":"code","source":"# Khai báo các hàm thực hiện tính trọng số\ncount_vectorizer = CountVectorizer()\ntf_idf_vectorizer = TfidfVectorizer()\n# Chia tập dữ liệu đầu vào thành 2 phần (tập train và test theo tỷ lệ (7:3))\nx_train, x_test, y_train, y_test = train_test_split(X, y, test_size=0.3)\n# Tiến hành tính toán trọng số của các từ trong tập huấn luyện\ncount_vectorizer.fit(x_train)\ntf_idf_vectorizer.fit(x_train)\n# Biến đổi các câu trong tập train thành ma trận trọng số cho 2 dạng vector hóa\nvt_count_train = count_vectorizer.transform(x_train)\nvt_count_test = count_vectorizer.transform(x_test)\n\nvt_tfidf_train = tf_idf_vectorizer.transform(x_train)\nvt_tfidf_test = tf_idf_vectorizer.transform(x_test)","metadata":{"execution":{"iopub.status.busy":"2021-12-21T13:59:05.012409Z","iopub.execute_input":"2021-12-21T13:59:05.013083Z","iopub.status.idle":"2021-12-21T13:59:54.108554Z","shell.execute_reply.started":"2021-12-21T13:59:05.013047Z","shell.execute_reply":"2021-12-21T13:59:54.107714Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":" ## Huấn luyện mô hình","metadata":{}},{"cell_type":"code","source":"# Khái báo mô hình\nmodel_countVectorizer = LogisticRegression(n_jobs=10, solver='saga', C=0.1, verbose=1)\nmodel_tfidfVectorizer = LogisticRegression(n_jobs=10, solver='saga', C=0.1, verbose=1)\n# Tiến hành huấn luyện trên tập dữ liệu đã được mã hóa bằng cả 2 phương pháp\nmodel_countVectorizer.fit(vt_count_train, y_train)\nmodel_tfidfVectorizer.fit(vt_count_train, y_train)\n\ny_pred_countVectorizer = model_countVectorizer.predict(vt_count_test)\ny_pred_tfidfVectorizer = model_tfidfVectorizer.predict(vt_tfidf_test)","metadata":{"execution":{"iopub.status.busy":"2021-12-21T13:59:54.110353Z","iopub.execute_input":"2021-12-21T13:59:54.110608Z","iopub.status.idle":"2021-12-21T14:01:04.386546Z","shell.execute_reply.started":"2021-12-21T13:59:54.110573Z","shell.execute_reply":"2021-12-21T14:01:04.385839Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Kết quả với Count Vectorizer","metadata":{}},{"cell_type":"code","source":"print(\"Logistic Regression with Count Vectorizer\\n\")\nprint('Recall: ', recall_score(y_pred_countVectorizer, y_test))\nprint('F1 score :', f1_score(y_pred_countVectorizer, y_test), '\\n')\nprint(classification_report(y_test, y_pred_countVectorizer))","metadata":{"execution":{"iopub.status.busy":"2021-12-21T14:01:04.388296Z","iopub.execute_input":"2021-12-21T14:01:04.388795Z","iopub.status.idle":"2021-12-21T14:01:04.997152Z","shell.execute_reply.started":"2021-12-21T14:01:04.388755Z","shell.execute_reply":"2021-12-21T14:01:04.995691Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Kết quả với TF-IDF","metadata":{}},{"cell_type":"code","source":"print(\"Logistic Regression with TF-IDF\\n\")\nprint('Recall: ', recall_score(y_pred_tfidfVectorizer, y_test))\nprint('F1 score :', f1_score(y_pred_tfidfVectorizer, y_test), '\\n')\nprint(classification_report(y_test, y_pred_tfidfVectorizer))","metadata":{"execution":{"iopub.status.busy":"2021-12-21T14:03:02.033576Z","iopub.execute_input":"2021-12-21T14:03:02.034382Z","iopub.status.idle":"2021-12-21T14:03:02.660852Z","shell.execute_reply.started":"2021-12-21T14:03:02.034346Z","shell.execute_reply":"2021-12-21T14:03:02.659961Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Kết luận:** Sử dụng vector hóa bằng phương pháp Count Vectorizer cho ra F1 score cao hơn nhiều so với việc sử dụng mã hóa bằng phương pháp TF_IDF.","metadata":{}},{"cell_type":"markdown","source":"# 3. Submission","metadata":{}},{"cell_type":"code","source":"test_data['clean_questions'] = test_data['question_text'].apply(clean)\nX_vec_test = count_vectorizer.transform(test_data['clean_questions'])\npredictions = model_tfidfVectorizer.predict(X_vec_test)\n","metadata":{"execution":{"iopub.status.busy":"2021-12-21T14:09:23.903587Z","iopub.execute_input":"2021-12-21T14:09:23.903840Z","iopub.status.idle":"2021-12-21T14:12:03.006049Z","shell.execute_reply.started":"2021-12-21T14:09:23.903813Z","shell.execute_reply":"2021-12-21T14:12:03.005300Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data['prediction'] = predictions\nresults = test_data[['qid', 'prediction']]\nprint(results)","metadata":{"execution":{"iopub.status.busy":"2021-12-21T14:14:35.465247Z","iopub.execute_input":"2021-12-21T14:14:35.465994Z","iopub.status.idle":"2021-12-21T14:14:35.483587Z","shell.execute_reply.started":"2021-12-21T14:14:35.465955Z","shell.execute_reply":"2021-12-21T14:14:35.482864Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"results.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2021-12-21T13:37:51.938521Z","iopub.execute_input":"2021-12-21T13:37:51.939343Z","iopub.status.idle":"2021-12-21T13:37:52.706238Z","shell.execute_reply.started":"2021-12-21T13:37:51.939305Z","shell.execute_reply":"2021-12-21T13:37:52.705526Z"},"trusted":true},"execution_count":null,"outputs":[]}]}