{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-06-11T00:41:02.408979Z","iopub.execute_input":"2021-06-11T00:41:02.409663Z","iopub.status.idle":"2021-06-11T00:41:02.426600Z","shell.execute_reply.started":"2021-06-11T00:41:02.409542Z","shell.execute_reply":"2021-06-11T00:41:02.424871Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"markdown","source":"<h3>Họ tên: Phương Anh Mỹ</h3>\n<h3>MSSV: 18020918</h3>","metadata":{}},{"cell_type":"markdown","source":"**1. Định nghĩa bài toán**\n<p>Một vấn đề tồn tại đối với bất kỳ trang web lớn nào hiện nay là làm thế nào để xử lý nội dung độc hại và gây chia rẽ. Quora muốn giải quyết vấn đề này trực tiếp để giữ cho nền tảng của họ trở thành một nơi mà người dùng có thể cảm thấy an toàn khi chia sẻ kiến thức của họ với cộng đồng.</p>\n<p>Quora là một nền tảng cho phép mọi người học hỏi và chia sẻ tri thức. Tại Quora, mọi người có thể đặt câu hỏi và kết nối với những người khác. Một thách thức lớn, đồng thời cũng là mục tiêu của bài toán, là xác định được những câu hỏi có nội dung nhạy cảm để có thể loại bỏ chúng.</p>\n<p>Input: Câu hỏi từ Quora</p>\n<p>Output: giá trị 0 hoặc 1 (0: câu hỏi không phản cảm; 1: câu hỏi phản cảm)</p>\n","metadata":{}},{"cell_type":"markdown","source":"**2. Dữ liệu**","metadata":{}},{"cell_type":"markdown","source":"2.1. Khảo sát dữ liệu","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport plotly.express as px\nimport plotly.graph_objects as go\nimport plotly.figure_factory as ff\nfrom plotly.subplots import make_subplots","metadata":{"execution":{"iopub.status.busy":"2021-06-11T00:41:02.428663Z","iopub.execute_input":"2021-06-11T00:41:02.429359Z","iopub.status.idle":"2021-06-11T00:41:03.892007Z","shell.execute_reply.started":"2021-06-11T00:41:02.429321Z","shell.execute_reply":"2021-06-11T00:41:03.890494Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df = pd.read_csv('../input/quora-insincere-questions-classification/test.csv')\ntrain_df = pd.read_csv('../input/quora-insincere-questions-classification/train.csv')","metadata":{"execution":{"iopub.status.busy":"2021-06-11T00:41:03.895014Z","iopub.execute_input":"2021-06-11T00:41:03.895497Z","iopub.status.idle":"2021-06-11T00:41:07.784222Z","shell.execute_reply.started":"2021-06-11T00:41:03.895446Z","shell.execute_reply":"2021-06-11T00:41:07.783424Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<p>Dữ liệu gồm 3 cột:</p>\n<p>qid: mã số câu hỏi</p>\n<p>question_text: câu hỏi</p> \n<p>target: nhãn dữ liệu (0 hoặc 1)</p>","metadata":{}},{"cell_type":"code","source":"train_df.info()","metadata":{"execution":{"iopub.status.busy":"2021-06-11T00:41:07.785718Z","iopub.execute_input":"2021-06-11T00:41:07.786269Z","iopub.status.idle":"2021-06-11T00:41:08.041168Z","shell.execute_reply.started":"2021-06-11T00:41:07.786235Z","shell.execute_reply":"2021-06-11T00:41:08.040289Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df.info()","metadata":{"execution":{"iopub.status.busy":"2021-06-11T00:41:08.042648Z","iopub.execute_input":"2021-06-11T00:41:08.043290Z","iopub.status.idle":"2021-06-11T00:41:08.126076Z","shell.execute_reply.started":"2021-06-11T00:41:08.043250Z","shell.execute_reply":"2021-06-11T00:41:08.124933Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Tập train gồm có 1306122 câu hỏi, tập test gồm có 375806 câu hỏi","metadata":{}},{"cell_type":"code","source":"sincere_question = train_df[train_df['target'] == 0].question_text #những câu hỏi không phản cảm thì sẽ có nhãn 0\ninsincere_question = train_df[train_df['target'] == 1].question_text #những câu hỏi không phản cảm thì sẽ có nhãn 1","metadata":{"execution":{"iopub.status.busy":"2021-06-11T00:41:08.127539Z","iopub.execute_input":"2021-06-11T00:41:08.127885Z","iopub.status.idle":"2021-06-11T00:41:08.238639Z","shell.execute_reply.started":"2021-06-11T00:41:08.127827Z","shell.execute_reply":"2021-06-11T00:41:08.237568Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* Trực quan hóa dữ liệu","metadata":{}},{"cell_type":"code","source":"sns.countplot(data=train_df, x='target')","metadata":{"execution":{"iopub.status.busy":"2021-06-11T00:41:08.240154Z","iopub.execute_input":"2021-06-11T00:41:08.240502Z","iopub.status.idle":"2021-06-11T00:41:08.478149Z","shell.execute_reply.started":"2021-06-11T00:41:08.240469Z","shell.execute_reply":"2021-06-11T00:41:08.476825Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":" print(\"Insincere questions: \", insincere_question.shape[0] / train_df.shape[0])\n print(\"Sincere questions: \", sincere_question.shape[0] / train_df.shape[0])","metadata":{"execution":{"iopub.status.busy":"2021-06-11T00:41:08.481089Z","iopub.execute_input":"2021-06-11T00:41:08.481418Z","iopub.status.idle":"2021-06-11T00:41:08.489068Z","shell.execute_reply.started":"2021-06-11T00:41:08.481387Z","shell.execute_reply":"2021-06-11T00:41:08.487577Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Có thể thấy, dữ liệu giữa 2 lớp bị mất cân bằng lớn (nhãn 0 chiếm 93,81% còn nhãn 1 chiếm 6.19%). Việc mất cân bằng dữ liệu sẽ gây ra khó khăn trong việc dự đoán lớp thiểu số. Vì vậy, ngoài việc huấn luyện mô hình, em sẽ giải quyết thêm vấn đề mất cân bằng dữ liệu để đạt được kết quả tốt hơn với phương pháp undersampling với tỉ lệ lớp 1:lớp 0 = 1:4.","metadata":{}},{"cell_type":"code","source":"sincere = train_df[train_df['target'] == 0]#những câu hỏi không phản cảm thì sẽ có nhãn 0\ninsincere = train_df[train_df['target'] == 1] #những câu hỏi không phản cảm thì sẽ có nhãn 1","metadata":{"execution":{"iopub.status.busy":"2021-06-11T00:41:08.491420Z","iopub.execute_input":"2021-06-11T00:41:08.491793Z","iopub.status.idle":"2021-06-11T00:41:08.592623Z","shell.execute_reply.started":"2021-06-11T00:41:08.491760Z","shell.execute_reply":"2021-06-11T00:41:08.591330Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.utils import resample\ndf_train_sampled = pd.concat([resample(sincere, replace = True, n_samples = len(insincere)*5), insincere])\ndf_train_sampled #dữ liệu đã được xử lý mất cân bằng","metadata":{"execution":{"iopub.status.busy":"2021-06-11T00:41:08.594051Z","iopub.execute_input":"2021-06-11T00:41:08.594395Z","iopub.status.idle":"2021-06-11T00:41:08.986600Z","shell.execute_reply.started":"2021-06-11T00:41:08.594351Z","shell.execute_reply":"2021-06-11T00:41:08.985778Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.countplot(data=df_train_sampled, x='target')","metadata":{"execution":{"iopub.status.busy":"2021-06-11T00:41:08.987714Z","iopub.execute_input":"2021-06-11T00:41:08.988014Z","iopub.status.idle":"2021-06-11T00:41:09.206603Z","shell.execute_reply.started":"2021-06-11T00:41:08.987985Z","shell.execute_reply":"2021-06-11T00:41:09.205203Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Một số câu hỏi phản cảm trong tập dữ liệu:","metadata":{}},{"cell_type":"code","source":"insincere_question.sample(n=5, random_state=4).values","metadata":{"execution":{"iopub.status.busy":"2021-06-11T00:41:09.208496Z","iopub.execute_input":"2021-06-11T00:41:09.208978Z","iopub.status.idle":"2021-06-11T00:41:09.220147Z","shell.execute_reply.started":"2021-06-11T00:41:09.208922Z","shell.execute_reply":"2021-06-11T00:41:09.218840Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Một số câu hỏi không phản cảm trong tập dữ liệu:","metadata":{}},{"cell_type":"code","source":"sincere_question.sample(n=5, random_state=4).values","metadata":{"execution":{"iopub.status.busy":"2021-06-11T00:41:09.221553Z","iopub.execute_input":"2021-06-11T00:41:09.221908Z","iopub.status.idle":"2021-06-11T00:41:09.303018Z","shell.execute_reply.started":"2021-06-11T00:41:09.221865Z","shell.execute_reply":"2021-06-11T00:41:09.301924Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"2.2. Tiền xử lý","metadata":{}},{"cell_type":"markdown","source":"Tiền xử lý dữ liệu bao gồm các việc: loại bỏ từ dừng, loại bỏ số, loại bỏ dấu câu, chuyển từ về dạng rút gọn.","metadata":{}},{"cell_type":"code","source":"contraction_dict = {\"ll\": \"will\", \"dont\": \"do not\", \"aint\": \"is not\", \"isnt\": \"is not\", \"doesnt\": \"does not\"\n, \"cant\": \"cannot\", \"mustnt\": \"must not\", \"hasnt\": \"has not\"\n, \"havent\": \"have not\", \"arent\": \"are not\", \"ain't\": \"is not\", \"aren't\": \"are not\"\n,\"can't\": \"cannot\", \"‘cause\": \"because\", \"could've\": \"could have\"\n, \"couldn't\": \"could not\", \"didn't\": \"did not\", \"doesn't\": \"does not\", \"don't\": \"do not\"\n, \"hadn't\": \"had not\", \"hasn't\": \"has not\", \"haven't\": \"have not\", \"he'd\": \"he would\"\n,\"he'll\": \"he will\", \"he's\": \"he is\", \"how'd\": \"how did\", \"how'd'y\": \"how do you\"\n, \"how'll\": \"how will\", \"how's\": \"how is\", \"I'd\": \"I would\", \"I'd've\": \"I would have\"\n, \"I'll\": \"I will\", \"I'll've\": \"I will have\",\"I'm\": \"I am\", \"Iam\": \"I am\", \"I've\": \"I have\"\n, \"i'd\": \"i would\", \"i'd've\": \"i would have\", \"i'll\": \"i will\", \"i'll've\": \"i will have\"\n,\"i'm\": \"i am\", \"i've\": \"i have\", \"isn't\": \"is not\", \"it'd\": \"it would\", \"it'd've\": \"it would have\"\n, \"it'll\": \"it will\", \"it'll've\": \"it will have\",\"it's\": \"it is\", \"let's\": \"let us\"\n, \"ma'am\": \"madam\", \"mayn't\": \"may not\", \"might've\": \"might have\"\n,\"mightn't\": \"might not\",\"mightn't've\": \"might not have\", \"must've\": \"must have\"\n, \"mustn't\": \"must not\", \"mustn't've\": \"must not have\", \"needn't\": \"need not\"\n, \"needn't've\": \"need not have\",\"o'clock\": \"of the clock\", \"oughtn't\": \"ought not\"\n, \"oughtn't've\": \"ought not have\", \"shan't\": \"shall not\", \"sha'n't\": \"shall not\"\n, \"shan't've\": \"shall not have\", \"she'd\": \"she would\", \"she'd've\": \"she would have\"\n, \"she'll\": \"she will\", \"she'll've\": \"she will have\", \"she's\": \"she is\", \n\"should've\": \"should have\", \"shouldn't\": \"should not\", \"shouldn't've\": \"should not have\"\n, \"so've\": \"so have\",\"so's\": \"so as\", \"this's\": \"this is\",\"that'd\": \"that would\", \n\"that'd've\": \"that would have\", \"that's\": \"that is\", \"there'd\": \"there would\", \n\"there'd've\": \"there would have\", \"there's\": \"there is\", \"here's\": \"here is\",\"they'd\": \"they would\"\n, \"they'd've\": \"they would have\", \"they'll\": \"they will\", \"they'll've\": \"they will have\", \n\"they're\": \"they are\", \"they've\": \"they have\", \"to've\": \"to have\", \"wasn't\": \"was not\",\n\"we'd\": \"we would\", \"we'd've\": \"we would have\", \"we'll\": \"we will\", \n\"we'll've\": \"we will have\", \"we're\": \"we are\", \"we've\": \"we have\", \n\"weren't\": \"were not\", \"what'll\": \"what will\", \"what'll've\": \"what will have\", \n\"what're\": \"what are\", \"what's\": \"what is\", \"what've\": \"what have\", \"when's\": \"when is\",\n\"when've\": \"when have\", \"where'd\": \"where did\", \"where's\": \"where is\", \"where've\": \"where have\", \n\"who'll\": \"who will\", \"who'll've\": \"who will have\", \"who's\": \"who is\", \"who've\": \"who have\",\n\"why's\": \"why is\", \"why've\": \"why have\", \"will've\": \"will have\", \"won't\": \"will not\", \"won't've\": \"will not have\",\n\"would've\": \"would have\", \"wouldn't\": \"would not\", \"wouldn't've\": \"would not have\", \"y'all\": \"you all\", \n\"y'all'd\": \"you all would\",\"y'all'd've\": \"you all would have\",\"y'all're\": \"you all are\",\n\"y'all've\": \"you all have\",\"you'd\": \"you would\", \"you'd've\": \"you would have\", \n\"you'll\": \"you will\", \"you'll've\": \"you will have\", \"you're\": \"you are\", \"you've\": \"you have\"}","metadata":{"execution":{"iopub.status.busy":"2021-06-11T00:41:09.304489Z","iopub.execute_input":"2021-06-11T00:41:09.304837Z","iopub.status.idle":"2021-06-11T00:41:09.321415Z","shell.execute_reply.started":"2021-06-11T00:41:09.304805Z","shell.execute_reply":"2021-06-11T00:41:09.319943Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import re\nimport nltk\nimport string\nfrom nltk.tokenize import word_tokenize\nfrom nltk.corpus import stopwords\nfrom nltk.stem import PorterStemmer, WordNetLemmatizer\nnltk.download('stopwords')\nlemmatizer = WordNetLemmatizer()\nstemmer = PorterStemmer()\nnltk_stopwords = stopwords.words('english')\ndef preprocessing(text):\n    # Data cleaning:\n    \n    text = re.sub('[0-9]{5,}','#####', text);\n    text = re.sub('[0-9]{4,}','####', text);\n    text = re.sub('[0-9]{3,}','###', text);\n    text = re.sub('[0-9]{2,}','##', text); \n    text = re.sub('[0-9]{1,}','#', text); \n    text = re.sub(re.compile('<.*?>'), '', text)\n    text = re.sub('[^A-Za-z0-9]+', ' ', text)\n    text = text.lower()\n\n    tokens = word_tokenize(text)\n    tokens = [contraction_dict.get(token) if (contraction_dict.get(token) != None) else token for token in tokens]\n    tokens = [w for w in tokens if w not in nltk_stopwords]\n    tokens = [stemmer.stem(token) for token in tokens]\n    tokens = [lemmatizer.lemmatize(w) for w in tokens]\n\n    # nối lại các từ vào chuỗi sau khi xử lý\n    text = ' '.join(tokens) \n\n    return text\n","metadata":{"execution":{"iopub.status.busy":"2021-06-11T00:41:09.323318Z","iopub.execute_input":"2021-06-11T00:41:09.323698Z","iopub.status.idle":"2021-06-11T00:41:10.177431Z","shell.execute_reply.started":"2021-06-11T00:41:09.323646Z","shell.execute_reply":"2021-06-11T00:41:10.175757Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Thực hiện công việc tiền xử lý với tập dữ liệu train trước khi xử lý mất cân bằng:","metadata":{}},{"cell_type":"code","source":"X_clean = []\nfor word in train_df.question_text:\n  X_clean.append(preprocessing(word))","metadata":{"execution":{"iopub.status.busy":"2021-06-11T00:41:10.179263Z","iopub.execute_input":"2021-06-11T00:41:10.179605Z","iopub.status.idle":"2021-06-11T00:51:23.577864Z","shell.execute_reply.started":"2021-06-11T00:41:10.179574Z","shell.execute_reply":"2021-06-11T00:51:23.576567Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Thực hiện công việc tiền xử lý với tập dữ liệu train sau khi xử lý mất cân bằng:","metadata":{}},{"cell_type":"code","source":"Y_clean = []\nfor word in df_train_sampled.question_text:\n  Y_clean.append(preprocessing(word))","metadata":{"execution":{"iopub.status.busy":"2021-06-11T00:51:23.579151Z","iopub.execute_input":"2021-06-11T00:51:23.579551Z","iopub.status.idle":"2021-06-11T00:55:20.076300Z","shell.execute_reply.started":"2021-06-11T00:51:23.579514Z","shell.execute_reply":"2021-06-11T00:55:20.075258Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Thêm trường câu hỏi đã được tiền xử lý đối với cả 2 tập dữ liệu:","metadata":{}},{"cell_type":"code","source":"train_df['cleaned_questions'] = X_clean\ntrain_df.to_csv('output_preprocessed.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2021-06-11T00:55:20.077577Z","iopub.execute_input":"2021-06-11T00:55:20.077897Z","iopub.status.idle":"2021-06-11T00:55:28.858137Z","shell.execute_reply.started":"2021-06-11T00:55:20.077867Z","shell.execute_reply":"2021-06-11T00:55:28.856974Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train_sampled['cleaned_questions'] = Y_clean\ndf_train_sampled.to_csv('Output_preprocessed.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2021-06-11T00:55:28.859385Z","iopub.execute_input":"2021-06-11T00:55:28.859690Z","iopub.status.idle":"2021-06-11T00:55:32.499163Z","shell.execute_reply.started":"2021-06-11T00:55:28.859661Z","shell.execute_reply":"2021-06-11T00:55:32.498152Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pre_data_before = pd.read_csv('output_preprocessed.csv')","metadata":{"execution":{"iopub.status.busy":"2021-06-11T00:55:32.500978Z","iopub.execute_input":"2021-06-11T00:55:32.501329Z","iopub.status.idle":"2021-06-11T00:55:36.676000Z","shell.execute_reply.started":"2021-06-11T00:55:32.501299Z","shell.execute_reply":"2021-06-11T00:55:36.674909Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preprocessed_data = pd.read_csv('Output_preprocessed.csv')","metadata":{"execution":{"iopub.status.busy":"2021-06-11T00:55:36.677393Z","iopub.execute_input":"2021-06-11T00:55:36.677671Z","iopub.status.idle":"2021-06-11T00:55:38.201466Z","shell.execute_reply.started":"2021-06-11T00:55:36.677644Z","shell.execute_reply":"2021-06-11T00:55:38.200277Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"2 tập dữ liệu đó có dạng sau:","metadata":{}},{"cell_type":"code","source":"train_df ","metadata":{"execution":{"iopub.status.busy":"2021-06-11T00:55:38.202943Z","iopub.execute_input":"2021-06-11T00:55:38.203278Z","iopub.status.idle":"2021-06-11T00:55:38.221548Z","shell.execute_reply.started":"2021-06-11T00:55:38.203243Z","shell.execute_reply":"2021-06-11T00:55:38.220552Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train_sampled","metadata":{"execution":{"iopub.status.busy":"2021-06-11T00:55:38.226650Z","iopub.execute_input":"2021-06-11T00:55:38.227013Z","iopub.status.idle":"2021-06-11T00:55:38.246986Z","shell.execute_reply.started":"2021-06-11T00:55:38.226981Z","shell.execute_reply":"2021-06-11T00:55:38.245777Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"markdown","source":"**3. Huấn luyện mô hình**","metadata":{}},{"cell_type":"markdown","source":"Mô hình sử dụng: Logistic Regression và Linear SVM. Ngoài ra, em có thử sử dụng mô hình Random Forest nhưng thời gian chạy khá lâu so với các mô hình học máy cơ bản khác.\nPhương pháp vector hóa dữ liệu: CountVectorizer và TF-IDFVectorizer (n-gram range: (1,2))\nChia dữ liệu để huấn luyện mô hình và dự đoán kết quả theo tỉ lệ test data:train data = 2:8. \nEm sẽ train với cả 2 bộ dữ liệu: trước khi xử lý mất cân bằng và sau khi xử lý mất cân bằng (tương đương với 8 lần train mô hình).","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split","metadata":{"execution":{"iopub.status.busy":"2021-06-11T00:55:38.249317Z","iopub.execute_input":"2021-06-11T00:55:38.249663Z","iopub.status.idle":"2021-06-11T00:55:38.260176Z","shell.execute_reply.started":"2021-06-11T00:55:38.249626Z","shell.execute_reply":"2021-06-11T00:55:38.259373Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Tập train, test trước khi xử lý mất cân bằng:","metadata":{}},{"cell_type":"code","source":"x, xtest, y, ytest = train_test_split(train_df['cleaned_questions'], train_df['target'], test_size = 0.2)","metadata":{"execution":{"iopub.status.busy":"2021-06-11T00:55:38.261692Z","iopub.execute_input":"2021-06-11T00:55:38.262131Z","iopub.status.idle":"2021-06-11T00:55:38.714251Z","shell.execute_reply.started":"2021-06-11T00:55:38.262092Z","shell.execute_reply":"2021-06-11T00:55:38.713040Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Tập train, test sau khi xử lý mất cân bằng:","metadata":{}},{"cell_type":"code","source":"X_train, X_test, y_train, y_test = train_test_split(df_train_sampled['cleaned_questions'], df_train_sampled['target'], test_size = 0.2)","metadata":{"execution":{"iopub.status.busy":"2021-06-11T00:55:38.716066Z","iopub.execute_input":"2021-06-11T00:55:38.716472Z","iopub.status.idle":"2021-06-11T00:55:38.806020Z","shell.execute_reply.started":"2021-06-11T00:55:38.716438Z","shell.execute_reply":"2021-06-11T00:55:38.804812Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Mô hình Logistic Regression**","metadata":{}},{"cell_type":"code","source":"from sklearn.linear_model import LogisticRegression\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.metrics import accuracy_score, f1_score\nfrom sklearn.metrics import classification_report\nfrom sklearn.feature_extraction.text import CountVectorizer\nfrom sklearn.feature_extraction.text import TfidfVectorizer\ncount_vectorizer = CountVectorizer(ngram_range=(1,2))\ntfidf_vectorizer = TfidfVectorizer(ngram_range=(1,2))\n","metadata":{"execution":{"iopub.status.busy":"2021-06-11T00:55:38.807566Z","iopub.execute_input":"2021-06-11T00:55:38.807940Z","iopub.status.idle":"2021-06-11T00:55:38.816589Z","shell.execute_reply.started":"2021-06-11T00:55:38.807903Z","shell.execute_reply":"2021-06-11T00:55:38.815294Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Train bằng phương pháp CountVectorizer với tập dữ liệu sau xử lý mất cân bằng:","metadata":{}},{"cell_type":"code","source":"model = LogisticRegression(C=1, random_state=0)\nvectorize_model_pipeline = Pipeline([\n    ('count_vectorizer', count_vectorizer),\n    ('model', model),\n])","metadata":{"execution":{"iopub.status.busy":"2021-06-11T00:55:38.818339Z","iopub.execute_input":"2021-06-11T00:55:38.818792Z","iopub.status.idle":"2021-06-11T00:55:38.833471Z","shell.execute_reply.started":"2021-06-11T00:55:38.818757Z","shell.execute_reply":"2021-06-11T00:55:38.832310Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"vectorize_model_pipeline.fit(X_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2021-06-11T00:55:38.835662Z","iopub.execute_input":"2021-06-11T00:55:38.836111Z","iopub.status.idle":"2021-06-11T00:56:38.385968Z","shell.execute_reply.started":"2021-06-11T00:55:38.836073Z","shell.execute_reply":"2021-06-11T00:56:38.384438Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred = vectorize_model_pipeline.predict(X_test)","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2021-06-11T00:56:38.387774Z","iopub.execute_input":"2021-06-11T00:56:38.388146Z","iopub.status.idle":"2021-06-11T00:56:40.781422Z","shell.execute_reply.started":"2021-06-11T00:56:38.388112Z","shell.execute_reply":"2021-06-11T00:56:40.780024Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Accuracy :', accuracy_score(y_test, y_pred))\nprint('F1 score :', f1_score(y_test, y_pred))","metadata":{"execution":{"iopub.status.busy":"2021-06-11T00:56:40.783180Z","iopub.execute_input":"2021-06-11T00:56:40.783703Z","iopub.status.idle":"2021-06-11T00:56:40.827777Z","shell.execute_reply.started":"2021-06-11T00:56:40.783664Z","shell.execute_reply":"2021-06-11T00:56:40.826540Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(classification_report(y_test, y_pred))","metadata":{"execution":{"iopub.status.busy":"2021-06-11T00:56:40.829226Z","iopub.execute_input":"2021-06-11T00:56:40.829519Z","iopub.status.idle":"2021-06-11T00:56:40.964317Z","shell.execute_reply.started":"2021-06-11T00:56:40.829490Z","shell.execute_reply":"2021-06-11T00:56:40.963030Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Train bằng phương pháp TF_IDFVectorizer với tập dữ liệu sau xử lý mất cân bằng:","metadata":{}},{"cell_type":"code","source":"vectorize_model_pipeline_ = Pipeline([\n    ('tfidf_vectorizer', tfidf_vectorizer),\n    ('model', model),\n])","metadata":{"execution":{"iopub.status.busy":"2021-06-11T00:56:40.965696Z","iopub.execute_input":"2021-06-11T00:56:40.966000Z","iopub.status.idle":"2021-06-11T00:56:40.973146Z","shell.execute_reply.started":"2021-06-11T00:56:40.965972Z","shell.execute_reply":"2021-06-11T00:56:40.971606Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"vectorize_model_pipeline_.fit(X_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2021-06-11T00:56:40.974722Z","iopub.execute_input":"2021-06-11T00:56:40.975170Z","iopub.status.idle":"2021-06-11T00:57:40.651446Z","shell.execute_reply.started":"2021-06-11T00:56:40.975137Z","shell.execute_reply":"2021-06-11T00:57:40.650442Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred_ = vectorize_model_pipeline_.predict(X_test)","metadata":{"execution":{"iopub.status.busy":"2021-06-11T00:57:40.653019Z","iopub.execute_input":"2021-06-11T00:57:40.653416Z","iopub.status.idle":"2021-06-11T00:57:43.082615Z","shell.execute_reply.started":"2021-06-11T00:57:40.653385Z","shell.execute_reply":"2021-06-11T00:57:43.081387Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(classification_report(y_test, y_pred_))","metadata":{"execution":{"iopub.status.busy":"2021-06-11T00:57:43.083890Z","iopub.execute_input":"2021-06-11T00:57:43.084172Z","iopub.status.idle":"2021-06-11T00:57:43.218561Z","shell.execute_reply.started":"2021-06-11T00:57:43.084145Z","shell.execute_reply":"2021-06-11T00:57:43.217298Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Train bằng phương pháp CountVectorizer với tập dữ liệu trước xử lý mất cân bằng:","metadata":{}},{"cell_type":"code","source":"vectorize_model_pipeline.fit(x, y)","metadata":{"execution":{"iopub.status.busy":"2021-06-11T00:57:43.219813Z","iopub.execute_input":"2021-06-11T00:57:43.220147Z","iopub.status.idle":"2021-06-11T01:00:00.779442Z","shell.execute_reply.started":"2021-06-11T00:57:43.220118Z","shell.execute_reply":"2021-06-11T01:00:00.777825Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred = vectorize_model_pipeline.predict(xtest)\nprint('Accuracy :', accuracy_score(ytest, y_pred))\nprint('F1 score :', f1_score(ytest, y_pred))\nprint(classification_report(ytest, y_pred))","metadata":{"execution":{"iopub.status.busy":"2021-06-11T01:00:00.781048Z","iopub.execute_input":"2021-06-11T01:00:00.781400Z","iopub.status.idle":"2021-06-11T01:00:07.366375Z","shell.execute_reply.started":"2021-06-11T01:00:00.781360Z","shell.execute_reply":"2021-06-11T01:00:07.365043Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Train bằng phương pháp TF_IDFVectorizer với tập dữ liệu trước xử lý mất cân bằng:","metadata":{}},{"cell_type":"code","source":"vectorize_model_pipeline_.fit(x, y)","metadata":{"execution":{"iopub.status.busy":"2021-06-11T01:00:07.368549Z","iopub.execute_input":"2021-06-11T01:00:07.368963Z","iopub.status.idle":"2021-06-11T01:02:20.284111Z","shell.execute_reply.started":"2021-06-11T01:00:07.368923Z","shell.execute_reply":"2021-06-11T01:02:20.282830Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred = vectorize_model_pipeline_.predict(xtest)\nprint('Accuracy :', accuracy_score(ytest, y_pred))\nprint('F1 score :', f1_score(ytest, y_pred))\nprint(classification_report(ytest, y_pred))","metadata":{"execution":{"iopub.status.busy":"2021-06-11T01:02:20.285531Z","iopub.execute_input":"2021-06-11T01:02:20.285827Z","iopub.status.idle":"2021-06-11T01:02:27.060057Z","shell.execute_reply.started":"2021-06-11T01:02:20.285799Z","shell.execute_reply":"2021-06-11T01:02:27.058471Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Có thể thấy rằng, độ đo F1 của mô hình (nhãn 1) với tập dữ liệu sau khi xử lý mất cân bằng tăng khá nhiều:\nCountVectorizer: 0.73 so với 0.53, TF_IDFVectorizer: 0.74 so với 0.54","metadata":{}},{"cell_type":"markdown","source":"**Mô hình Linear SVM**","metadata":{}},{"cell_type":"markdown","source":"Train bằng phương pháp CounerVectorizer với tập dữ liệu sau xử lý mất cân bằng:","metadata":{}},{"cell_type":"code","source":"from sklearn.svm import LinearSVC\nmodel1 = LinearSVC(random_state=3, tol=0.01, loss='hinge', C=1, verbose=2)\nvectorize_model_pipeline1 = Pipeline([\n    ('count_vectorizer', count_vectorizer),\n    ('model1', model1),\n])\nvectorize_model_pipeline1.fit(X_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2021-06-11T01:02:27.061440Z","iopub.execute_input":"2021-06-11T01:02:27.061780Z","iopub.status.idle":"2021-06-11T01:03:04.147615Z","shell.execute_reply.started":"2021-06-11T01:02:27.061720Z","shell.execute_reply":"2021-06-11T01:03:04.146470Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred1 = vectorize_model_pipeline1.predict(X_test)\nprint('Accuracy :', accuracy_score(y_test, y_pred1))\nprint('F1 score :', f1_score(y_test, y_pred1))\nprint(classification_report(y_test, y_pred1))","metadata":{"execution":{"iopub.status.busy":"2021-06-11T01:03:04.151372Z","iopub.execute_input":"2021-06-11T01:03:04.151672Z","iopub.status.idle":"2021-06-11T01:03:06.680107Z","shell.execute_reply.started":"2021-06-11T01:03:04.151642Z","shell.execute_reply":"2021-06-11T01:03:06.679330Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Train bằng phương pháp TF_IDFVectorizer với tập dữ liệu sau xử lý mất cân bằng:","metadata":{}},{"cell_type":"code","source":"vectorize_model_pipeline1_ = Pipeline([\n    ('count_vectorizer', tfidf_vectorizer),\n    ('model1', model1),\n])\nvectorize_model_pipeline1_.fit(X_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2021-06-11T01:03:06.681114Z","iopub.execute_input":"2021-06-11T01:03:06.681510Z","iopub.status.idle":"2021-06-11T01:03:26.244167Z","shell.execute_reply.started":"2021-06-11T01:03:06.681481Z","shell.execute_reply":"2021-06-11T01:03:26.243052Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred1 = vectorize_model_pipeline1_.predict(X_test)\nprint('Accuracy :', accuracy_score(y_test, y_pred1))\nprint('F1 score :', f1_score(y_test, y_pred1))\nprint(classification_report(y_test, y_pred1))","metadata":{"execution":{"iopub.status.busy":"2021-06-11T01:03:26.245429Z","iopub.execute_input":"2021-06-11T01:03:26.245745Z","iopub.status.idle":"2021-06-11T01:03:28.842466Z","shell.execute_reply.started":"2021-06-11T01:03:26.245713Z","shell.execute_reply":"2021-06-11T01:03:28.841007Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Train bằng phương pháp CountVectorizer với tập dữ liệu trước xử lý mất cân bằng:","metadata":{}},{"cell_type":"code","source":"vectorize_model_pipeline1.fit(x, y)\ny_pred1 = vectorize_model_pipeline1.predict(xtest)\nprint('Accuracy :', accuracy_score(ytest, y_pred1))\nprint('F1 score :', f1_score(ytest, y_pred1))\nprint(classification_report(ytest, y_pred1))","metadata":{"execution":{"iopub.status.busy":"2021-06-11T01:03:28.845082Z","iopub.execute_input":"2021-06-11T01:03:28.845535Z","iopub.status.idle":"2021-06-11T01:05:15.716383Z","shell.execute_reply.started":"2021-06-11T01:03:28.845485Z","shell.execute_reply":"2021-06-11T01:05:15.714755Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Train bằng phương pháp TF-IDFVectorizer với tập dữ liệu trước xử lý mất cân bằng:","metadata":{}},{"cell_type":"code","source":"vectorize_model_pipeline1_.fit(x, y)\ny_pred1 = vectorize_model_pipeline1_.predict(xtest)\nprint('Accuracy :', accuracy_score(ytest, y_pred1))\nprint('F1 score :', f1_score(ytest, y_pred1))\nprint(classification_report(ytest, y_pred1))","metadata":{"execution":{"iopub.status.busy":"2021-06-11T01:05:15.717788Z","iopub.execute_input":"2021-06-11T01:05:15.718132Z","iopub.status.idle":"2021-06-11T01:06:16.239923Z","shell.execute_reply.started":"2021-06-11T01:05:15.718101Z","shell.execute_reply":"2021-06-11T01:06:16.238895Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Có thể thấy rằng, độ đo F1 của mô hình (nhãn 1) với tập dữ liệu sau khi xử lý mất cân bằng tăng khá nhiều:\nCountVectorizer: 0.74 so với 0.54, TF_IDFVectorizer: 0.76 so với 0.56","metadata":{}},{"cell_type":"code","source":"# from sklearn.metrics import classification_report\n# print(classification_report(y_test, predictions1))","metadata":{"execution":{"iopub.status.busy":"2021-06-11T01:06:16.241205Z","iopub.execute_input":"2021-06-11T01:06:16.241507Z","iopub.status.idle":"2021-06-11T01:06:16.245623Z","shell.execute_reply.started":"2021-06-11T01:06:16.241476Z","shell.execute_reply":"2021-06-11T01:06:16.244539Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.ensemble import RandomForestClassifier\ncount_vectorizer = CountVectorizer()\nmodel2 = RandomForestClassifier()\n\nvectorize_model_pipeline2 = Pipeline([\n    ('count_vectorizer', count_vectorizer),\n    ('model2', model2)])\nvectorize_model_pipeline2.fit(X_train, y_train)\npredictions2 = vectorize_model_pipeline2.predict(X_test)\n\nprint('Accuracy :', accuracy_score(y_test, predictions2))\nprint('F1 score :', accuracy_score(y_test, predictions2))","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(classification_report(y_test, predictions2))","metadata":{"execution":{"iopub.status.busy":"2021-06-11T01:09:15.469720Z","iopub.status.idle":"2021-06-11T01:09:15.470196Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Nộp kết quả**","metadata":{}},{"cell_type":"markdown","source":"Các lần submit cho thấy mô hình Logistic Regession với CountVectorizer cho kết quả tốt nhất.","metadata":{}},{"cell_type":"code","source":"test_df['preprocessing'] = test_df['question_text'].apply(preprocessing) #tiền xử lý với dữ liệu test","metadata":{"execution":{"iopub.status.busy":"2021-06-11T01:09:40.834637Z","iopub.execute_input":"2021-06-11T01:09:40.835006Z","iopub.status.idle":"2021-06-11T01:12:37.852282Z","shell.execute_reply.started":"2021-06-11T01:09:40.834975Z","shell.execute_reply":"2021-06-11T01:12:37.851135Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions = vectorize_model_pipeline.predict(test_df['preprocessing']) #dự đoán kết quả","metadata":{"execution":{"iopub.status.busy":"2021-06-11T01:12:37.854069Z","iopub.execute_input":"2021-06-11T01:12:37.854376Z","iopub.status.idle":"2021-06-11T01:12:46.773483Z","shell.execute_reply.started":"2021-06-11T01:12:37.854347Z","shell.execute_reply":"2021-06-11T01:12:46.772070Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df['prediction'] = predictions\nresults = test_df[['qid', 'prediction']]\nresults.to_csv('submission.csv', index=False) #lưu vào file submission","metadata":{"execution":{"iopub.status.busy":"2021-06-11T01:12:46.775399Z","iopub.execute_input":"2021-06-11T01:12:46.775719Z","iopub.status.idle":"2021-06-11T01:12:47.697810Z","shell.execute_reply.started":"2021-06-11T01:12:46.775689Z","shell.execute_reply":"2021-06-11T01:12:47.696740Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"results.head()","metadata":{"execution":{"iopub.status.busy":"2021-06-11T01:12:47.701094Z","iopub.execute_input":"2021-06-11T01:12:47.701668Z","iopub.status.idle":"2021-06-11T01:12:47.712114Z","shell.execute_reply.started":"2021-06-11T01:12:47.701620Z","shell.execute_reply":"2021-06-11T01:12:47.711138Z"},"trusted":true},"execution_count":null,"outputs":[]}]}