{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport random\nfrom sklearn.model_selection import train_test_split\nimport re\nfrom tqdm import tqdm\nprint(\"Setup Done\")","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-06-10T16:42:20.792085Z","iopub.execute_input":"2021-06-10T16:42:20.792537Z","iopub.status.idle":"2021-06-10T16:42:20.799500Z","shell.execute_reply.started":"2021-06-10T16:42:20.792498Z","shell.execute_reply":"2021-06-10T16:42:20.798174Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Mô tả bài toán\n\nCho các dữ liệu về các câu hỏi trên Quora dạng text và phân loại của chúng(Có phải toxic hay không?)\n\nĐánh giá xem dữ liệu cho có phải là câu hỏi toxic hay không?","metadata":{}},{"cell_type":"code","source":"raw_train_data = pd.read_csv('/kaggle/input/quora-insincere-questions-classification/train.csv')\nraw_test_data = pd.read_csv('/kaggle/input/quora-insincere-questions-classification/test.csv')\n#đọc file dữ liệu\nraw_train_data.head()","metadata":{"execution":{"iopub.status.busy":"2021-06-10T16:42:20.801471Z","iopub.execute_input":"2021-06-10T16:42:20.801908Z","iopub.status.idle":"2021-06-10T16:42:26.049496Z","shell.execute_reply.started":"2021-06-10T16:42:20.801858Z","shell.execute_reply":"2021-06-10T16:42:26.048651Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"raw_test_data.head()","metadata":{"execution":{"iopub.status.busy":"2021-06-10T16:42:26.051122Z","iopub.execute_input":"2021-06-10T16:42:26.051551Z","iopub.status.idle":"2021-06-10T16:42:26.063993Z","shell.execute_reply.started":"2021-06-10T16:42:26.051518Z","shell.execute_reply":"2021-06-10T16:42:26.062842Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Tiền xử lý dữ liệu\n> # -Các công việc cần thiết\n> Trước tiên,cần xử lý qua dữ liệu text này trước.Những việc cần phải làm những việc sau:\n> > * **Sơ chế dữ liệu thô.**\n> > * **Chia tập train và tập validation.**\n> > * **Loại bỏ các nhiễu.**\n> > * **Tạo vector feature cho dữ liệu**","metadata":{}},{"cell_type":"markdown","source":"****","metadata":{}},{"cell_type":"markdown","source":"> # -Sơ chế dữ liệu thô\n> \n> Có 3 cột ở đây\n> > * Id of question(Text)\n> > * Question text(Text)\n> > * Is question is sincere or not?(0/1)","metadata":{}},{"cell_type":"code","source":"feature_name = ['question_text','target']\n#lấy dữ liệu ở 2 cột question_text và target\ntrain_data = raw_train_data[feature_name]\n#tạo ra mảng mới có 2 cột dữ liệu question_text và target\nprint(train_data.head())\n#in kết quả ra theo định dạng","metadata":{"execution":{"iopub.status.busy":"2021-06-10T16:42:26.065647Z","iopub.execute_input":"2021-06-10T16:42:26.065970Z","iopub.status.idle":"2021-06-10T16:42:26.293488Z","shell.execute_reply.started":"2021-06-10T16:42:26.065939Z","shell.execute_reply":"2021-06-10T16:42:26.292077Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"> # -Chia tập train và tập validation\n> * Phải tách tập train và tập validation ra sớm,tránh tập validation có thể bị ô nhiễm bởi tập train trong quá trình tiền xử lý dữ liệu","metadata":{}},{"cell_type":"code","source":"### chia ra thành 2 tập,tập train và tập validation\ntrain,val=train_test_split(train_data,test_size=0.2,stratify=train_data.target,random_state=123)\n#Random 80% là mảng train và 20% là mảng validation\nprint(\"Shape of the Training set :\",train.shape)\n#in ra kích thức của mảng dữ liệu Train\nprint(\"Shape of the Validation set :\",val.shape)\n#in ra kích thức của mảng dữ liệu Validation\nprint(train.head())\n#in dữ liệu ra theo định dạng của mảng Train\nprint(train.question_text[224356])\n#in dữ liệu của question có id là 224356","metadata":{"execution":{"iopub.status.busy":"2021-06-10T16:42:26.294979Z","iopub.execute_input":"2021-06-10T16:42:26.295573Z","iopub.status.idle":"2021-06-10T16:42:27.608779Z","shell.execute_reply.started":"2021-06-10T16:42:26.295533Z","shell.execute_reply":"2021-06-10T16:42:27.607544Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* Test thử dữ liệu vừa tách","metadata":{}},{"cell_type":"code","source":"#print(train.columns)\n#print(train.question_text[602217])\n#for i in range(0,10,1):\n#    randIndex = random.randrange(0,train.question_text.size,1)\n#   print(train.question_text[randIndex])","metadata":{"execution":{"iopub.status.busy":"2021-06-10T16:42:27.609954Z","iopub.execute_input":"2021-06-10T16:42:27.610253Z","iopub.status.idle":"2021-06-10T16:42:27.614624Z","shell.execute_reply.started":"2021-06-10T16:42:27.610225Z","shell.execute_reply":"2021-06-10T16:42:27.613467Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"\n phải reset_index cho dữ liệu mới để đồng bộ.","metadata":{}},{"cell_type":"code","source":" #sắp xếp lại các cột chỉ số\ntrain = train.reset_index()\nval = val.reset_index()\n#in ra 10 gtri test thử\nprint(train.head(10))","metadata":{"execution":{"iopub.status.busy":"2021-06-10T16:42:27.615962Z","iopub.execute_input":"2021-06-10T16:42:27.616289Z","iopub.status.idle":"2021-06-10T16:42:27.715429Z","shell.execute_reply.started":"2021-06-10T16:42:27.616261Z","shell.execute_reply":"2021-06-10T16:42:27.714339Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"> # -Loại bỏ các nhiễu\n> Khi xử lý dữ liệu,cần lọc ra những dữ liệu cần thiết và không cần thiết cho việc dự đoán đánh giá của model.Trong bài này,dữ liệu cần là các từ,nên các kí tự đặc biệt và số sẽ bị loại bỏ.\n> > * Loại bỏ các từ viết tắt\n> > * Loại bỏ các kí tự đặc biệt,không có trong bảng chữ cái","metadata":{}},{"cell_type":"markdown","source":"* **Loại bỏ các từ viết tắt**","metadata":{}},{"cell_type":"code","source":"#Loại bỏ các từ viết tắt \n\n#Tạo một directory có key là các từ viết tắt và value là các từ không viết tắt của key\n#Mục đích để thay thế các từ viết tắt bằng các từ chuẩn,để chuẩn bị tạo danh sách từ vựng\ncontractions={\"I'm\": 'I am',\n \"I'm'a\": 'I am about to',\n \"I'm'o\": 'I am going to',\n \"I've\": 'I have',\n \"I'll\": 'I will',\n \"I'll've\": 'I will have',\n \"I'd\": 'I would',\n \"I'd've\": 'I would have',\n 'Whatcha': 'What are you',\n \"amn't\": 'am not',\n \"ain't\": 'are not',\n \"aren't\": 'are not',\n \"'cause\": 'because',\n \"can't\": 'can not',\n \"can't've\": 'can not have',\n \"could've\": 'could have',\n \"couldn't\": 'could not',\n \"couldn't've\": 'could not have',\n \"daren't\": 'dare not',\n \"daresn't\": 'dare not',\n \"dasn't\": 'dare not',\n \"didn't\": 'did not',\n 'didn’t': 'did not',\n \"don't\": 'do not',\n 'don’t': 'do not',\n \"doesn't\": 'does not',\n \"e'er\": 'ever',\n \"everyone's\": 'everyone is',\n 'finna': 'fixing to',\n 'gimme': 'give me',\n \"gon't\": 'go not',\n 'gonna': 'going to',\n 'gotta': 'got to',\n \"hadn't\": 'had not',\n \"hadn't've\": 'had not have',\n \"hasn't\": 'has not',\n \"haven't\": 'have not',\n \"he've\": 'he have',\n \"he's\": 'he is',\n \"he'll\": 'he will',\n \"he'll've\": 'he will have',\n \"he'd\": 'he would',\n \"he'd've\": 'he would have',\n \"here's\": 'here is',\n \"how're\": 'how are',\n \"how'd\": 'how did',\n \"how'd'y\": 'how do you',\n \"how's\": 'how is',\n \"how'll\": 'how will',\n \"isn't\": 'is not',\n \"it's\": 'it is',\n \"'tis\": 'it is',\n \"'twas\": 'it was',\n \"it'll\": 'it will',\n \"it'll've\": 'it will have',\n \"it'd\": 'it would',\n \"it'd've\": 'it would have',\n 'kinda': 'kind of',\n \"let's\": 'let us',\n 'luv': 'love',\n \"ma'am\": 'madam',\n \"may've\": 'may have',\n \"mayn't\": 'may not',\n \"might've\": 'might have',\n \"mightn't\": 'might not',\n \"mightn't've\": 'might not have',\n \"must've\": 'must have',\n \"mustn't\": 'must not',\n \"mustn't've\": 'must not have',\n \"needn't\": 'need not',\n \"needn't've\": 'need not have',\n \"ne'er\": 'never',\n \"o'\": 'of',\n \"o'clock\": 'of the clock',\n \"ol'\": 'old',\n \"oughtn't\": 'ought not',\n \"oughtn't've\": 'ought not have',\n \"o'er\": 'over',\n \"shan't\": 'shall not',\n \"sha'n't\": 'shall not',\n \"shalln't\": 'shall not',\n \"shan't've\": 'shall not have',\n \"she's\": 'she is',\n \"she'll\": 'she will',\n \"she'd\": 'she would',\n \"she'd've\": 'she would have',\n \"should've\": 'should have',\n \"shouldn't\": 'should not',\n \"shouldn't've\": 'should not have',\n \"so've\": 'so have',\n \"so's\": 'so is',\n \"somebody's\": 'somebody is',\n \"someone's\": 'someone is',\n \"something's\": 'something is',\n 'sux': 'sucks',\n \"that're\": 'that are',\n \"that's\": 'that is',\n \"that'll\": 'that will',\n \"that'd\": 'that would',\n \"that'd've\": 'that would have',\n 'em': 'them',\n \"there're\": 'there are',\n \"there's\": 'there is',\n \"there'll\": 'there will',\n \"there'd\": 'there would',\n \"there'd've\": 'there would have',\n \"these're\": 'these are',\n \"they're\": 'they are',\n \"they've\": 'they have',\n \"they'll\": 'they will',\n \"they'll've\": 'they will have',\n \"they'd\": 'they would',\n \"they'd've\": 'they would have',\n \"this's\": 'this is',\n \"those're\": 'those are',\n \"to've\": 'to have',\n 'wanna': 'want to',\n \"wasn't\": 'was not',\n \"we're\": 'we are',\n \"we've\": 'we have',\n \"we'll\": 'we will',\n \"we'll've\": 'we will have',\n \"we'd\": 'we would',\n \"we'd've\": 'we would have',\n \"weren't\": 'were not',\n \"what're\": 'what are',\n \"what'd\": 'what did',\n \"what've\": 'what have',\n \"what's\": 'what is',\n \"what'll\": 'what will',\n \"what'll've\": 'what will have',\n \"when've\": 'when have',\n \"when's\": 'when is',\n \"where're\": 'where are',\n \"where'd\": 'where did',\n \"where've\": 'where have',\n \"where's\": 'where is',\n \"which's\": 'which is',\n \"who're\": 'who are',\n \"who've\": 'who have',\n \"who's\": 'who is',\n \"who'll\": 'who will',\n \"who'll've\": 'who will have',\n \"who'd\": 'who would',\n \"who'd've\": 'who would have',\n \"why're\": 'why are',\n \"why'd\": 'why did',\n \"why've\": 'why have',\n \"why's\": 'why is',\n \"will've\": 'will have',\n \"won't\": 'will not',\n \"won't've\": 'will not have',\n \"would've\": 'would have',\n \"wouldn't\": 'would not',\n \"wouldn't've\": 'would not have',\n \"y'all\": 'you all',\n \"y'all're\": 'you all are',\n \"y'all've\": 'you all have',\n \"y'all'd\": 'you all would',\n \"y'all'd've\": 'you all would have',\n \"you're\": 'you are',\n \"you've\": 'you have',\n \"you'll've\": 'you shall have',\n \"you'll\": 'you will',\n \"you'd\": 'you would',\n \"you'd've\": 'you would have',\n 'jan.': 'january',\n 'feb.': 'february',\n 'mar.': 'march',\n 'apr.': 'april',\n 'jun.': 'june',\n 'jul.': 'july',\n 'aug.': 'august',\n 'sep.': 'september',\n 'oct.': 'october',\n 'nov.': 'november',\n 'dec.': 'december',\n 'I’m': 'I am',\n 'I’m’a': 'I am about to',\n 'I’m’o': 'I am going to',\n 'I’ve': 'I have',\n 'I’ll': 'I will',\n 'I’ll’ve': 'I will have',\n 'I’d': 'I would',\n 'I’d’ve': 'I would have',\n 'amn’t': 'am not',\n 'ain’t': 'are not',\n 'aren’t': 'are not',\n '’cause': 'because',\n 'can’t': 'can not',\n 'can’t’ve': 'can not have',\n 'could’ve': 'could have',\n 'couldn’t': 'could not',\n 'couldn’t’ve': 'could not have',\n 'daren’t': 'dare not',\n 'daresn’t': 'dare not',\n 'dasn’t': 'dare not',\n 'doesn’t': 'does not',\n 'e’er': 'ever',\n 'everyone’s': 'everyone is',\n 'gon’t': 'go not',\n 'hadn’t': 'had not',\n 'hadn’t’ve': 'had not have',\n 'hasn’t': 'has not',\n 'haven’t': 'have not',\n 'he’ve': 'he have',\n 'he’s': 'he is',\n 'he’ll': 'he will',\n 'he’ll’ve': 'he will have',\n 'he’d': 'he would',\n 'he’d’ve': 'he would have',\n 'here’s': 'here is',\n 'how’re': 'how are',\n 'how’d': 'how did',\n 'how’d’y': 'how do you',\n 'how’s': 'how is',\n 'how’ll': 'how will',\n 'isn’t': 'is not',\n 'it’s': 'it is',\n '’tis': 'it is',\n '’twas': 'it was',\n 'it’ll': 'it will',\n 'it’ll’ve': 'it will have',\n 'it’d': 'it would',\n 'it’d’ve': 'it would have',\n 'let’s': 'let us',\n 'ma’am': 'madam',\n 'may’ve': 'may have',\n 'mayn’t': 'may not',\n 'might’ve': 'might have',\n 'mightn’t': 'might not',\n 'mightn’t’ve': 'might not have',\n 'must’ve': 'must have',\n 'mustn’t': 'must not',\n 'mustn’t’ve': 'must not have',\n 'needn’t': 'need not',\n 'needn’t’ve': 'need not have',\n 'ne’er': 'never',\n 'o’': 'of',\n 'o’clock': 'of the clock',\n 'ol’': 'old',\n 'oughtn’t': 'ought not',\n 'oughtn’t’ve': 'ought not have',\n 'o’er': 'over',\n 'shan’t': 'shall not',\n 'sha’n’t': 'shall not',\n 'shalln’t': 'shall not',\n 'shan’t’ve': 'shall not have',\n 'she’s': 'she is',\n 'she’ll': 'she will',\n 'she’d': 'she would',\n 'she’d’ve': 'she would have',\n 'should’ve': 'should have',\n 'shouldn’t': 'should not',\n 'shouldn’t’ve': 'should not have',\n 'so’ve': 'so have',\n 'so’s': 'so is',\n 'somebody’s': 'somebody is',\n 'someone’s': 'someone is',\n 'something’s': 'something is',\n 'that’re': 'that are',\n 'that’s': 'that is',\n 'that’ll': 'that will',\n 'that’d': 'that would',\n 'that’d’ve': 'that would have',\n 'there’re': 'there are',\n 'there’s': 'there is',\n 'there’ll': 'there will',\n 'there’d': 'there would',\n 'there’d’ve': 'there would have',\n 'these’re': 'these are',\n 'they’re': 'they are',\n 'they’ve': 'they have',\n 'they’ll': 'they will',\n 'they’ll’ve': 'they will have',\n 'they’d': 'they would',\n 'they’d’ve': 'they would have',\n 'this’s': 'this is',\n 'those’re': 'those are',\n 'to’ve': 'to have',\n 'wasn’t': 'was not',\n 'we’re': 'we are',\n 'we’ve': 'we have',\n 'we’ll': 'we will',\n 'we’ll’ve': 'we will have',\n 'we’d': 'we would',\n 'we’d’ve': 'we would have',\n 'weren’t': 'were not',\n 'what’re': 'what are',\n 'what’d': 'what did',\n 'what’ve': 'what have',\n 'what’s': 'what is',\n 'what’ll': 'what will',\n 'what’ll’ve': 'what will have',\n 'when’ve': 'when have',\n 'when’s': 'when is',\n 'where’re': 'where are',\n 'where’d': 'where did',\n 'where’ve': 'where have',\n 'where’s': 'where is',\n 'which’s': 'which is',\n 'who’re': 'who are',\n 'who’ve': 'who have',\n 'who’s': 'who is',\n 'who’ll': 'who will',\n 'who’ll’ve': 'who will have',\n 'who’d': 'who would',\n 'who’d’ve': 'who would have',\n 'why’re': 'why are',\n 'why’d': 'why did',\n 'why’ve': 'why have',\n 'why’s': 'why is',\n 'will’ve': 'will have',\n 'won’t': 'will not',\n 'won’t’ve': 'will not have',\n 'would’ve': 'would have',\n 'wouldn’t': 'would not',\n 'wouldn’t’ve': 'would not have',\n 'y’all': 'you all',\n 'y’all’re': 'you all are',\n 'y’all’ve': 'you all have',\n 'y’all’d': 'you all would',\n 'y’all’d’ve': 'you all would have',\n 'you’re': 'you are',\n 'you’ve': 'you have',\n 'you’ll’ve': 'you shall have',\n 'you’ll': 'you will',\n 'you’d': 'you would',\n 'you’d’ve': 'you would have'}\n\n#Hàm chuyển đổi các từ viết tắt thành các cụm từ chuẩn,không viết tắt\ndef contraction_fix(word):\n    try:\n        a=contractions[word]#nếu word là từ viết tắt có trong bộ từ viết tắt => a sẽ là cụm từ không viết tắt của word\n    except KeyError:\n        a=word # nếu không có key nào trong directory phù hợp với word đã cho=> a sẽ vẫn là word\n    return a #trả về từ vựng(Cụm từ vựng không viết tắt)","metadata":{"execution":{"iopub.status.busy":"2021-06-10T16:42:27.718194Z","iopub.execute_input":"2021-06-10T16:42:27.718474Z","iopub.status.idle":"2021-06-10T16:42:27.756169Z","shell.execute_reply.started":"2021-06-10T16:42:27.718446Z","shell.execute_reply":"2021-06-10T16:42:27.755112Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* **Loại bỏ các chữ số,các kí tự đặc biệt**","metadata":{}},{"cell_type":"code","source":"###Loại bỏ các chữ số,các kí tự đặc biệt\n\ndef Preprocess(doc): #Hàm loại bỏ các chữ số,các kí tự đặc biệt\n    corpus=[]\n    for text in tqdm(doc):\n        text=\" \".join([contraction_fix(w) for w in text.split()])   #tách các từ,thay thế các từ viết tắt bằng các từ đúng,sau đó nối chúng lại\n        \n        #re là một module để xác định biểu thức chính quy(là một đoạn các ký tự đặc biệt dùng để so khớp các chuỗi hoặc một tập các chuỗi)\n        text=re.sub(r'[^a-z0-9A-Z]',\" \",text)#Loại bỏ các dấu như !,?.... thay bằng các ' '\n        text=re.sub(r'[0-9]{1}',\"#\",text)#Loại bỏ các số,thay bằng các '#'\n        text=re.sub(r'[0-9]{2}','##',text)\n        text=re.sub(r'[0-9]{3}','###',text)\n        text=re.sub(r'[0-9]{4}','####',text)\n        text=re.sub(r'[0-9]{5,}','#####',text)\n        corpus.append(text) #thêm dòng vừa rồi vào mảng corpus\n    \n    return corpus #trả về mảng các câu đã được lược bỏ số và các kí tự đặc biệt","metadata":{"execution":{"iopub.status.busy":"2021-06-10T16:42:27.758064Z","iopub.execute_input":"2021-06-10T16:42:27.758428Z","iopub.status.idle":"2021-06-10T16:42:27.774548Z","shell.execute_reply.started":"2021-06-10T16:42:27.758397Z","shell.execute_reply":"2021-06-10T16:42:27.773607Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"> # -Tạo các vector feature từ dữ liệu\n> \n> Khi đã xử lý qua được dữ liệu,cần tiếp tục chuyển đổi dữ liệu vừa được xử lý đó sang dữ liệu có thể training cho model\n> > * Lấy ra từ vựng cho các câu\n> > * Kiến tạo vector cho các dữ liệu","metadata":{}},{"cell_type":"markdown","source":"* **Lấy ra từ vựng cho các câu**\n","metadata":{}},{"cell_type":"code","source":"###Sau khi đã có hàm tiền xử lý cần thiết,tạo hàm lấy ra vốn từ vựng\n\ndef get_vocab(corpus):\n    vocab={}#Đây là vốn từ vựng của chúng ta(Directory)\n    for text in tqdm(corpus): #Lặp qua các câu trong danh sách câu đã được xử lý\n        for word in text.split(): #Lặp qua các từ trong các câu\n            try:\n                vocab[word]+=1 #Nếu từ đó đã có rồi,+1 thêm vào số lượng của từ đó\n            except KeyError:\n                vocab[word]=1 #Nếu chưa có từ đó,tạo ra 1 key = word và value = 1 mới trong vocab(Directory)\n    vocab=dict(sorted(vocab.items(),reverse=True ,key=lambda item: item[1]))#Sắp kếp lại các Key\n    return vocab","metadata":{"execution":{"iopub.status.busy":"2021-06-10T16:42:27.775791Z","iopub.execute_input":"2021-06-10T16:42:27.776131Z","iopub.status.idle":"2021-06-10T16:42:27.793521Z","shell.execute_reply.started":"2021-06-10T16:42:27.776099Z","shell.execute_reply":"2021-06-10T16:42:27.792345Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"    Thử hàm tiền xử lý","metadata":{}},{"cell_type":"code","source":"#chạy hàm loại bỏ kí tự đặc biệt phía trên\ntrain_processed_doc = Preprocess(train.question_text)\nval_processed_doc = Preprocess(val.question_text)","metadata":{"execution":{"iopub.status.busy":"2021-06-10T16:42:27.794948Z","iopub.execute_input":"2021-06-10T16:42:27.795285Z","iopub.status.idle":"2021-06-10T16:43:19.003289Z","shell.execute_reply.started":"2021-06-10T16:42:27.795255Z","shell.execute_reply":"2021-06-10T16:43:19.002240Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in range(0,10,1):\n    randIndex = random.randrange(0,train.question_text.size,1)\n    print(\"Raw:\" + train.question_text[randIndex])\n    #in ra question ban đầu\n    print(\"Process:\" + train_processed_doc[randIndex])\n    #in ra hàm tiền sử lý","metadata":{"execution":{"iopub.status.busy":"2021-06-10T16:43:19.004876Z","iopub.execute_input":"2021-06-10T16:43:19.005266Z","iopub.status.idle":"2021-06-10T16:43:19.013506Z","shell.execute_reply.started":"2021-06-10T16:43:19.005233Z","shell.execute_reply":"2021-06-10T16:43:19.011825Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"    Hàm tiền xử lý đã ổn","metadata":{}},{"cell_type":"markdown","source":"    Tiến hành lấy ra các từ vựng","metadata":{}},{"cell_type":"code","source":"vocabulary = get_vocab(train_processed_doc)","metadata":{"execution":{"iopub.status.busy":"2021-06-10T16:43:19.015179Z","iopub.execute_input":"2021-06-10T16:43:19.015487Z","iopub.status.idle":"2021-06-10T16:43:24.020350Z","shell.execute_reply.started":"2021-06-10T16:43:19.015449Z","shell.execute_reply":"2021-06-10T16:43:24.019148Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Test thử lấy độ dài của danh sách vốn từ\nlen(vocabulary)","metadata":{"execution":{"iopub.status.busy":"2021-06-10T16:43:24.021774Z","iopub.execute_input":"2021-06-10T16:43:24.022197Z","iopub.status.idle":"2021-06-10T16:43:24.028703Z","shell.execute_reply.started":"2021-06-10T16:43:24.022153Z","shell.execute_reply":"2021-06-10T16:43:24.027637Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":" có 199881 từ","metadata":{}},{"cell_type":"code","source":"#vocabulary.items()\nprint(\"Bỏ comment nếu cần xem qua các item\")","metadata":{"execution":{"iopub.status.busy":"2021-06-10T16:43:24.030261Z","iopub.execute_input":"2021-06-10T16:43:24.030672Z","iopub.status.idle":"2021-06-10T16:43:24.040983Z","shell.execute_reply.started":"2021-06-10T16:43:24.030627Z","shell.execute_reply":"2021-06-10T16:43:24.039832Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* **Tạo vector**\n    * Xử lý theo cách nguyên thủy:\n\n        * Tạo ra một mảng có kích thước bằng số lượng từ vựng để chứa dữ liệu","metadata":{}},{"cell_type":"code","source":"##Import\nfrom sklearn.naive_bayes import MultinomialNB, BernoulliNB\nfrom scipy.sparse import coo_matrix\nfrom sklearn.metrics import accuracy_score ","metadata":{"execution":{"iopub.status.busy":"2021-06-10T16:43:24.042520Z","iopub.execute_input":"2021-06-10T16:43:24.042987Z","iopub.status.idle":"2021-06-10T16:43:24.052518Z","shell.execute_reply.started":"2021-06-10T16:43:24.042945Z","shell.execute_reply":"2021-06-10T16:43:24.051675Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"##thử xử lý dữ liệu theo kiểu nguyên thủy nhất\ndef naiveProcessData(raw_data):\n    processedData = []\n    for data in tqdm(raw_data):\n        #tạo mới directory với các key là vocabulary và value = 0;\n        processedSentence = vocabulary.copy()\n        for key in dictOfVocabulary.keys():\n            processedSentence[key] = 0\n        \n        #cộng 1 với mỗi word có trong câu.\n        for word in data.split():\n            try:\n                processedSentence[word] += 1;    \n            except:\n                print(word + \"is not in bag of word\")\n        processedData.append(np.array(processedSentence.values()))\n    return processedData","metadata":{"execution":{"iopub.status.busy":"2021-06-10T16:43:24.054217Z","iopub.execute_input":"2021-06-10T16:43:24.054641Z","iopub.status.idle":"2021-06-10T16:43:24.063554Z","shell.execute_reply.started":"2021-06-10T16:43:24.054599Z","shell.execute_reply":"2021-06-10T16:43:24.062550Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* **Tạo vector**\n\n    * Xử lý theo ma trận thưa\n","metadata":{}},{"cell_type":"code","source":"###Tạo hàm khởi tạo các giá trị cần thiết cho ma trận thưa\n\n#đánh dấu số thứ tự vị trí của các từ trước\nindex_Vocabulary = vocabulary.copy()\ndef initIndexVocabulary(indexVocabulary):\n    i = 0;\n    for key in indexVocabulary.keys():\n        indexVocabulary[key] = i\n        i += 1","metadata":{"execution":{"iopub.status.busy":"2021-06-10T16:43:24.065418Z","iopub.execute_input":"2021-06-10T16:43:24.065820Z","iopub.status.idle":"2021-06-10T16:43:24.098850Z","shell.execute_reply.started":"2021-06-10T16:43:24.065779Z","shell.execute_reply":"2021-06-10T16:43:24.097930Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"initIndexVocabulary(index_Vocabulary)\n#print(index_Vocabulary.items())","metadata":{"execution":{"iopub.status.busy":"2021-06-10T16:43:24.100146Z","iopub.execute_input":"2021-06-10T16:43:24.100441Z","iopub.status.idle":"2021-06-10T16:43:24.162939Z","shell.execute_reply.started":"2021-06-10T16:43:24.100412Z","shell.execute_reply":"2021-06-10T16:43:24.162010Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"##Tạo một mảng dữ liệu 3 giá trị như sau: Hàng(dòng dữ liệu thứ mấy),cột(từ đó là từ thứ bao nhiêu trong vocabulary),giá trị(từ đó xuất hiện bao nhiêu lần)\ndef createConfigValueForMatixSprase(doc):\n    rows = []\n    columns = []\n    values = []\n    \n    indexSentence = 0; # index của dữ liệu câu hỏi\n    for sentence in tqdm(doc): #lặp qua từng câu hỏi trong kho dữ liệu\n        dictOfWord = {}  #Khởi tạo tập hợp các từ có trong câu\n        for word in sentence.split(): #Xét các từ bên trong một câu\n            try: # Nếu tập hợp đã có từ đang xét,số lượng từ đó trong dict tăng lên 1\n                dictOfWord[word] += 1;\n            except: # Nếu tập hợp chưa có từ đang xét,thêm từ đó vào trong dict,với số lượng từ bằng 1\n                dictOfWord.update({word:1})\n        #Lặp qua tất cả các từ có trong từ điển của câu.\n        for word in dictOfWord.keys():\n            rows.append(indexSentence) #Thêm dữ liệu thứ tự dòng\n            try:\n                columns.append(index_Vocabulary[word]) #Thêm dữ liệu thứ thự cột(Vị trí của từ trong index_Vocabulary)\n            except:\n                columns.append(len(index_Vocabulary)) # Nếu từ đó không có trong vocabulary,cho nó xuống hang cuối cùng\n            values.append(dictOfWord[word]) # Thêm dữ liệu về số lần xuất hiện của từ đó\n        indexSentence += 1;\n    return (rows,columns,values)","metadata":{"execution":{"iopub.status.busy":"2021-06-10T16:43:24.164084Z","iopub.execute_input":"2021-06-10T16:43:24.164350Z","iopub.status.idle":"2021-06-10T16:43:24.171947Z","shell.execute_reply.started":"2021-06-10T16:43:24.164326Z","shell.execute_reply":"2021-06-10T16:43:24.171188Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"##Tạo ma trận thưa \ndef createSparseMatrix(data):\n    rows,columns,values = createConfigValueForMatixSprase(data)\n    return coo_matrix((values,(rows,columns)),shape=(len(data),len(index_Vocabulary) + 1))","metadata":{"execution":{"iopub.status.busy":"2021-06-10T16:43:24.173262Z","iopub.execute_input":"2021-06-10T16:43:24.173523Z","iopub.status.idle":"2021-06-10T16:43:24.188700Z","shell.execute_reply.started":"2021-06-10T16:43:24.173498Z","shell.execute_reply":"2021-06-10T16:43:24.187702Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Khởi tạo ma trận traning\ntrain_X = createSparseMatrix(train_processed_doc) #tạo ma trận thưa cho tập train\nlabel_X = train.target","metadata":{"execution":{"iopub.status.busy":"2021-06-10T16:43:24.190060Z","iopub.execute_input":"2021-06-10T16:43:24.190366Z","iopub.status.idle":"2021-06-10T16:43:51.656938Z","shell.execute_reply.started":"2021-06-10T16:43:24.190339Z","shell.execute_reply":"2021-06-10T16:43:51.656143Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_y = createSparseMatrix(val_processed_doc) #tạo ma trận thưa cho tập validation\nvalid_y = val.target","metadata":{"execution":{"iopub.status.busy":"2021-06-10T16:43:51.661342Z","iopub.execute_input":"2021-06-10T16:43:51.661907Z","iopub.status.idle":"2021-06-10T16:43:58.508608Z","shell.execute_reply.started":"2021-06-10T16:43:51.661856Z","shell.execute_reply":"2021-06-10T16:43:58.507693Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Training Model\n> * Chọn model\n> * Training và thử nghiệm","metadata":{}},{"cell_type":"markdown","source":"> # -Chọn model\n\n> đây là bài toán phân loại và là bài toán về ngôn ngữ,nên sử dụng mô hình Naive Bayes Classifiers.\n\n> Việc xử lý dữ liệu đầu vào cho model đã được giải quyết bằng cách xử dụng ma trận thưa ở trên phần tạo vector feature cho dữ liệu","metadata":{}},{"cell_type":"markdown","source":"> # -Training Model","metadata":{}},{"cell_type":"markdown","source":"* **Sử dụng MultinomialNB**","metadata":{}},{"cell_type":"code","source":"#Training Model MultinomialNB()\nmulti_NB_model = MultinomialNB()\nmulti_NB_model.fit(train_X,label_X)","metadata":{"execution":{"iopub.status.busy":"2021-06-10T16:43:58.510238Z","iopub.execute_input":"2021-06-10T16:43:58.510572Z","iopub.status.idle":"2021-06-10T16:43:59.222862Z","shell.execute_reply.started":"2021-06-10T16:43:58.510543Z","shell.execute_reply":"2021-06-10T16:43:59.221717Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* **Tạo các hàm thử đánh giá**","metadata":{}},{"cell_type":"code","source":"#Đánh giá model qua tập validation\ndef validModel(Model):\n    pred_y = Model.predict(test_y)\n    print('Training size = %d, accuracy = %.2f%%' % \\\n          (train_X.shape[0],accuracy_score(valid_y, pred_y)*100))","metadata":{"execution":{"iopub.status.busy":"2021-06-10T16:43:59.224099Z","iopub.execute_input":"2021-06-10T16:43:59.224422Z","iopub.status.idle":"2021-06-10T16:43:59.230135Z","shell.execute_reply.started":"2021-06-10T16:43:59.224390Z","shell.execute_reply":"2021-06-10T16:43:59.229049Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def compareModel(Models):\n    indexRand = random.randrange(0,len(val.target) - 20,1) #chọn random 1 giá trị\n    test_dataframe = val[indexRand:indexRand + 10] #lấy 10 giá trị liên tiếp từ số mới random được\n    test_data = test_dataframe.question_text\n    processed_test_data = Preprocess(test_data) #gọi hàm loại bỏ các kí tự đặc biệt\n    test_data_X = createSparseMatrix(processed_test_data) #tạo ma trận thưa với giữ liệu vừa được xử lý\n    \n    \n    for model in Models:\n        y_valid_test = test_dataframe.target\n        y_pred_test = model.predict(test_data_X)\n        print(type(model).__name__[0:4]+\":\" + str(list(y_pred_test)))\n    print(\"test:\" + str(list(y_valid_test))) #in ra kết quả test","metadata":{"execution":{"iopub.status.busy":"2021-06-10T16:43:59.231492Z","iopub.execute_input":"2021-06-10T16:43:59.231789Z","iopub.status.idle":"2021-06-10T16:43:59.243262Z","shell.execute_reply.started":"2021-06-10T16:43:59.231758Z","shell.execute_reply":"2021-06-10T16:43:59.242101Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Hàm thử nghiệm 1 chút\ndef testModel(Model):\n    indexRand = random.randrange(0,len(val.target) - 20,1) #random lấy 1 giá trị\n    test_dataframe = val[indexRand:indexRand + 10] #lấy 10 giá trị liên tiếp từ số mới random được\n    test_data = test_dataframe.question_text\n    processed_test_data = Preprocess(test_data) #gọi hàm loại bỏ các kí tự đặc biệt\n    test_data_X = createSparseMatrix(processed_test_data) #tạo ma trận thưa với giữ liệu vừa được xử lý\n\n\n    y_valid_test = test_dataframe.target\n    y_pred_test = Model.predict(test_data_X)\n    print(test_dataframe)\n    print(\"test:\" + str(list(y_valid_test))) #in ra kết quả test\n    print(\"pred:\" + str(list(y_pred_test))) #in ra kết quả thực tế","metadata":{"execution":{"iopub.status.busy":"2021-06-10T16:43:59.244592Z","iopub.execute_input":"2021-06-10T16:43:59.244904Z","iopub.status.idle":"2021-06-10T16:43:59.254704Z","shell.execute_reply.started":"2021-06-10T16:43:59.244875Z","shell.execute_reply":"2021-06-10T16:43:59.253694Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* **Thử nghiệm độ chính xác của model thông qua validation**","metadata":{}},{"cell_type":"code","source":"#đánh giá MultinomialNB()\nvalidModel(multi_NB_model)","metadata":{"execution":{"iopub.status.busy":"2021-06-10T16:43:59.256860Z","iopub.execute_input":"2021-06-10T16:43:59.257916Z","iopub.status.idle":"2021-06-10T16:43:59.431018Z","shell.execute_reply.started":"2021-06-10T16:43:59.257860Z","shell.execute_reply":"2021-06-10T16:43:59.429834Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"    Độ chính xác hơn 92%","metadata":{}},{"cell_type":"code","source":"testModel(multi_NB_model)","metadata":{"execution":{"iopub.status.busy":"2021-06-10T16:43:59.432848Z","iopub.execute_input":"2021-06-10T16:43:59.433379Z","iopub.status.idle":"2021-06-10T16:43:59.455290Z","shell.execute_reply.started":"2021-06-10T16:43:59.433326Z","shell.execute_reply":"2021-06-10T16:43:59.454177Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"    Thử nghiệm khá là chuẩn","metadata":{}},{"cell_type":"markdown","source":"* **sử dụng BernoulliNB**","metadata":{}},{"cell_type":"code","source":"ber_NB_model = BernoulliNB()\nber_NB_model.fit(train_X,label_X)","metadata":{"execution":{"iopub.status.busy":"2021-06-10T16:43:59.456901Z","iopub.execute_input":"2021-06-10T16:43:59.457564Z","iopub.status.idle":"2021-06-10T16:44:00.249003Z","shell.execute_reply.started":"2021-06-10T16:43:59.457515Z","shell.execute_reply":"2021-06-10T16:44:00.247959Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"validModel(ber_NB_model)","metadata":{"execution":{"iopub.status.busy":"2021-06-10T16:44:00.250453Z","iopub.execute_input":"2021-06-10T16:44:00.250774Z","iopub.status.idle":"2021-06-10T16:44:00.444692Z","shell.execute_reply.started":"2021-06-10T16:44:00.250742Z","shell.execute_reply":"2021-06-10T16:44:00.443368Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"    Độ chính xác hơi tốt hơn một chút","metadata":{}},{"cell_type":"code","source":"testModel(ber_NB_model)","metadata":{"execution":{"iopub.status.busy":"2021-06-10T16:44:00.445954Z","iopub.execute_input":"2021-06-10T16:44:00.446306Z","iopub.status.idle":"2021-06-10T16:44:00.477405Z","shell.execute_reply.started":"2021-06-10T16:44:00.446273Z","shell.execute_reply":"2021-06-10T16:44:00.476489Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"> # -Training model với toàn bộ dữ liệu train\nchọn ber_NB_model vì chính xác hơn 1 chút","metadata":{}},{"cell_type":"code","source":"all_train_preprocess_data = Preprocess(train_data.question_text)\nall_train_x = createSparseMatrix(all_train_preprocess_data)\nall_train_y = raw_train_data.target\nber_NB_model.fit(all_train_x,all_train_y)","metadata":{"execution":{"iopub.status.busy":"2021-06-10T16:44:00.478669Z","iopub.execute_input":"2021-06-10T16:44:00.478955Z","iopub.status.idle":"2021-06-10T16:45:25.984433Z","shell.execute_reply.started":"2021-06-10T16:44:00.478926Z","shell.execute_reply":"2021-06-10T16:45:25.983353Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"validModel(ber_NB_model)","metadata":{"execution":{"iopub.status.busy":"2021-06-10T16:45:25.986833Z","iopub.execute_input":"2021-06-10T16:45:25.987132Z","iopub.status.idle":"2021-06-10T16:45:26.180847Z","shell.execute_reply.started":"2021-06-10T16:45:25.987105Z","shell.execute_reply":"2021-06-10T16:45:26.179773Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"train toàn bộ thì được kết quả tốt hơn chút nữa","metadata":{}},{"cell_type":"markdown","source":"> # -Vector hóa dữ liệu cần predict","metadata":{}},{"cell_type":"code","source":"raw_test_data.head()\npreprocessTestData = Preprocess(raw_test_data.question_text) #lấy tập đã loại bỏ hết dữ liệu\ntestt_padded = createSparseMatrix(preprocessTestData) #đưa dữ liệu vừa lấy được thành ma trận thưa","metadata":{"execution":{"iopub.status.busy":"2021-06-10T16:45:26.182156Z","iopub.execute_input":"2021-06-10T16:45:26.182439Z","iopub.status.idle":"2021-06-10T16:45:50.247394Z","shell.execute_reply.started":"2021-06-10T16:45:26.182412Z","shell.execute_reply":"2021-06-10T16:45:50.246320Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"> # -Predict dữ liệu","metadata":{}},{"cell_type":"code","source":"y_test_pre = ber_NB_model.predict(testt_padded)\n#print(y_test_pre)","metadata":{"execution":{"iopub.status.busy":"2021-06-10T16:45:50.248572Z","iopub.execute_input":"2021-06-10T16:45:50.248858Z","iopub.status.idle":"2021-06-10T16:45:50.480194Z","shell.execute_reply.started":"2021-06-10T16:45:50.248831Z","shell.execute_reply":"2021-06-10T16:45:50.479176Z"},"trusted":true},"execution_count":null,"outputs":[]}]}