{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Import các thư viện và load data","metadata":{}},{"cell_type":"markdown","source":"Notebook <br>\nHọ và tên: Trần Ngọc Hướng <br>\nMssv: 19021297 <br>\nHọc phần Học máy <br>","metadata":{}},{"cell_type":"markdown","source":"Import thư viện","metadata":{}},{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n%matplotlib inline\nimport matplotlib as mp \nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom wordcloud import WordCloud","metadata":{"execution":{"iopub.status.busy":"2022-01-08T14:51:17.005073Z","iopub.execute_input":"2022-01-08T14:51:17.005612Z","iopub.status.idle":"2022-01-08T14:51:17.808045Z","shell.execute_reply.started":"2022-01-08T14:51:17.005529Z","shell.execute_reply":"2022-01-08T14:51:17.807171Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Loading data","metadata":{}},{"cell_type":"code","source":"train_data = pd.read_csv('/kaggle/input/quora-insincere-questions-classification/train.csv')\ntest_data = pd.read_csv('/kaggle/input/quora-insincere-questions-classification/test.csv')\nprint(\"train_data\",train_data.shape)\nprint(\"test_data\",test_data.shape)","metadata":{"execution":{"iopub.status.busy":"2022-01-08T14:51:17.810289Z","iopub.execute_input":"2022-01-08T14:51:17.810585Z","iopub.status.idle":"2022-01-08T14:51:23.757438Z","shell.execute_reply.started":"2022-01-08T14:51:17.810558Z","shell.execute_reply":"2022-01-08T14:51:23.756131Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Phân tích dữ liệu","metadata":{}},{"cell_type":"markdown","source":"In ra một vài dữ liệu ở tập huấn luyện","metadata":{}},{"cell_type":"code","source":"train_data.head(5)","metadata":{"execution":{"iopub.status.busy":"2022-01-08T14:51:23.759189Z","iopub.execute_input":"2022-01-08T14:51:23.759589Z","iopub.status.idle":"2022-01-08T14:51:23.789981Z","shell.execute_reply.started":"2022-01-08T14:51:23.759547Z","shell.execute_reply":"2022-01-08T14:51:23.788632Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.target.value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-01-08T14:51:23.791839Z","iopub.execute_input":"2022-01-08T14:51:23.792395Z","iopub.status.idle":"2022-01-08T14:51:23.815048Z","shell.execute_reply.started":"2022-01-08T14:51:23.792349Z","shell.execute_reply":"2022-01-08T14:51:23.814202Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Biểu đồ target 0 và 1","metadata":{}},{"cell_type":"code","source":"sns.countplot(data=train_data, x='target')","metadata":{"execution":{"iopub.status.busy":"2022-01-08T14:51:23.819066Z","iopub.execute_input":"2022-01-08T14:51:23.819467Z","iopub.status.idle":"2022-01-08T14:51:24.077576Z","shell.execute_reply.started":"2022-01-08T14:51:23.819422Z","shell.execute_reply":"2022-01-08T14:51:24.076636Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"In ra một vài dữ liệu ở tập train","metadata":{}},{"cell_type":"code","source":"test_data.head(5)","metadata":{"execution":{"iopub.status.busy":"2022-01-08T14:51:24.081543Z","iopub.execute_input":"2022-01-08T14:51:24.081875Z","iopub.status.idle":"2022-01-08T14:51:24.091427Z","shell.execute_reply.started":"2022-01-08T14:51:24.081844Z","shell.execute_reply":"2022-01-08T14:51:24.090602Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"In ra một vài câu hỏi chân thành và không chân thành","metadata":{}},{"cell_type":"code","source":"print(train_data['question_text'][(train_data['target']==0)].sample(5).values)","metadata":{"execution":{"iopub.status.busy":"2022-01-08T14:51:24.092544Z","iopub.execute_input":"2022-01-08T14:51:24.092861Z","iopub.status.idle":"2022-01-08T14:51:24.271034Z","shell.execute_reply.started":"2022-01-08T14:51:24.092833Z","shell.execute_reply":"2022-01-08T14:51:24.269968Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(train_data['question_text'][(train_data['target']==1)].sample(5).values)","metadata":{"execution":{"iopub.status.busy":"2022-01-08T14:51:24.272220Z","iopub.execute_input":"2022-01-08T14:51:24.272659Z","iopub.status.idle":"2022-01-08T14:51:24.295406Z","shell.execute_reply.started":"2022-01-08T14:51:24.272624Z","shell.execute_reply":"2022-01-08T14:51:24.294551Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Vẽ word cloud trực quan hóa dữ liệu","metadata":{}},{"cell_type":"code","source":"print(\"Các từ thường xuất hiện trong những câu hỏi không chân thành\")\ninsincere_ques_text =\" \".join(train_data[train_data[\"target\"] == 1][\"question_text\"])\n# Create the wordcloud object\nwordcloud = WordCloud(width=600, height=600, margin=0).generate(insincere_ques_text)\n# Display the generated image:\nfig,ax = plt.subplots(1,1,figsize=(10,10))\nax.imshow(wordcloud, interpolation='bilinear')\nax.axis(\"off\")\nax.margins(x = 0, y = 0)\nplt.show() ","metadata":{"execution":{"iopub.status.busy":"2022-01-08T14:51:24.296850Z","iopub.execute_input":"2022-01-08T14:51:24.297138Z","iopub.status.idle":"2022-01-08T14:51:30.961315Z","shell.execute_reply.started":"2022-01-08T14:51:24.297111Z","shell.execute_reply":"2022-01-08T14:51:30.960399Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Xử lý dữ liệu","metadata":{}},{"cell_type":"code","source":"import nltk\nimport sys\nimport spacy\nimport string\nfrom unidecode import unidecode\nimport re\nfrom nltk.stem import PorterStemmer, WordNetLemmatizer\nfrom nltk.tokenize import word_tokenize\nfrom nltk.corpus import stopwords","metadata":{"execution":{"iopub.status.busy":"2022-01-08T14:51:30.962372Z","iopub.execute_input":"2022-01-08T14:51:30.962797Z","iopub.status.idle":"2022-01-08T14:51:32.302446Z","shell.execute_reply.started":"2022-01-08T14:51:30.962756Z","shell.execute_reply":"2022-01-08T14:51:32.301522Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Hàm chuyển chữ viết tắt thành chữ thường\ncontraction_mapping = {\"ain't\": \"is not\", \"aren't\": \"are not\",\"can't\": \"cannot\", \"'cause\": \"because\", \"could've\": \"could have\", \"couldn't\": \"could not\", \"didn't\": \"did not\",  \"doesn't\": \"does not\", \"don't\": \"do not\", \"hadn't\": \"had not\", \"hasn't\": \"has not\", \"haven't\": \"have not\", \"he'd\": \"he would\",\"he'll\": \"he will\", \"he's\": \"he is\", \"how'd\": \"how did\", \"how'd'y\": \"how do you\", \"how'll\": \"how will\", \"how's\": \"how is\",  \"I'd\": \"I would\", \"I'd've\": \"I would have\", \"I'll\": \"I will\", \"I'll've\": \"I will have\",\"I'm\": \"I am\", \"I've\": \"I have\", \"i'd\": \"i would\", \"i'd've\": \"i would have\", \"i'll\": \"i will\",  \"i'll've\": \"i will have\",\"i'm\": \"i am\", \"i've\": \"i have\", \"isn't\": \"is not\", \"it'd\": \"it would\", \"it'd've\": \"it would have\", \"it'll\": \"it will\", \"it'll've\": \"it will have\",\"it's\": \"it is\", \"let's\": \"let us\", \"ma'am\": \"madam\", \"mayn't\": \"may not\", \"might've\": \"might have\",\"mightn't\": \"might not\",\"mightn't've\": \"might not have\", \"must've\": \"must have\", \"mustn't\": \"must not\", \"mustn't've\": \"must not have\", \"needn't\": \"need not\", \"needn't've\": \"need not have\",\"o'clock\": \"of the clock\", \"oughtn't\": \"ought not\", \"oughtn't've\": \"ought not have\", \"shan't\": \"shall not\", \"sha'n't\": \"shall not\", \"shan't've\": \"shall not have\", \"she'd\": \"she would\", \"she'd've\": \"she would have\", \"she'll\": \"she will\", \"she'll've\": \"she will have\", \"she's\": \"she is\", \"should've\": \"should have\", \"shouldn't\": \"should not\", \"shouldn't've\": \"should not have\", \"so've\": \"so have\",\"so's\": \"so as\", \"this's\": \"this is\",\"that'd\": \"that would\", \"that'd've\": \"that would have\", \"that's\": \"that is\", \"there'd\": \"there would\", \"there'd've\": \"there would have\", \"there's\": \"there is\", \"here's\": \"here is\",\"they'd\": \"they would\", \"they'd've\": \"they would have\", \"they'll\": \"they will\", \"they'll've\": \"they will have\", \"they're\": \"they are\", \"they've\": \"they have\", \"to've\": \"to have\", \"wasn't\": \"was not\", \"we'd\": \"we would\", \"we'd've\": \"we would have\", \"we'll\": \"we will\", \"we'll've\": \"we will have\", \"we're\": \"we are\", \"we've\": \"we have\", \"weren't\": \"were not\", \"what'll\": \"what will\", \"what'll've\": \"what will have\", \"what're\": \"what are\",  \"what's\": \"what is\", \"what've\": \"what have\", \"when's\": \"when is\", \"when've\": \"when have\", \"where'd\": \"where did\", \"where's\": \"where is\", \"where've\": \"where have\", \"who'll\": \"who will\", \"who'll've\": \"who will have\", \"who's\": \"who is\", \"who've\": \"who have\", \"why's\": \"why is\", \"why've\": \"why have\", \"will've\": \"will have\", \"won't\": \"will not\", \"won't've\": \"will not have\", \"would've\": \"would have\", \"wouldn't\": \"would not\", \"wouldn't've\": \"would not have\", \"y'all\": \"you all\", \"y'all'd\": \"you all would\",\"y'all'd've\": \"you all would have\",\"y'all're\": \"you all are\",\"y'all've\": \"you all have\",\"you'd\": \"you would\", \"you'd've\": \"you would have\", \"you'll\": \"you will\", \"you'll've\": \"you will have\", \"you're\": \"you are\", \"you've\": \"you have\" }\ndef clean_contractions(text, mapping):\n    specials = [\"’\", \"‘\", \"´\", \"`\"]\n    for s in specials:\n        text = text.replace(s, \"'\")\n    text = ' '.join([mapping[t] if t in mapping else t for t in text.split(\" \")])\n    return text","metadata":{"execution":{"iopub.status.busy":"2022-01-08T14:51:32.303974Z","iopub.execute_input":"2022-01-08T14:51:32.304267Z","iopub.status.idle":"2022-01-08T14:51:32.323732Z","shell.execute_reply.started":"2022-01-08T14:51:32.304238Z","shell.execute_reply":"2022-01-08T14:51:32.322762Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#nguồn: https://www.kaggle.com/canming/ensemble-mean-iii-64-36\n#Hàm clean tag toán học (math) và URL\ndef clean_tag(text):\n    if '[math]' in text:\n        text = re.sub('\\[math\\].*?math\\]', '[formula]', text)\n    if 'http' in text or 'www' in text:\n        text = re.sub('(?:(?:https?|ftp):\\/\\/)?[\\w/\\-?=%.]+\\.[\\w/\\-?=%.]+', '[url]', text)\n    return text","metadata":{"execution":{"iopub.status.busy":"2022-01-08T14:51:32.325470Z","iopub.execute_input":"2022-01-08T14:51:32.325856Z","iopub.status.idle":"2022-01-08T14:51:32.338894Z","shell.execute_reply.started":"2022-01-08T14:51:32.325818Z","shell.execute_reply":"2022-01-08T14:51:32.337822Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Dấu và các ký tự đặc biệt\npuncts=[',', '.', '\"', ':', ')', '(', '-', '!', '?', '|', ';', \"'\", '$', '&', '/', '[', ']', '>', '%', '=', '#', '*', '+', '\\\\', \n        '•', '~', '@', '£', '·', '_', '{', '}', '©', '^', '®', '`', '<', '→', '°', '€', '™', '›', '♥', '←', '×', '§', '″', '′', \n        '█', '…', '“', '★', '”', '–', '●', '►', '−', '¢', '¬', '░', '¡', '¶', '↑', '±', '¿', '▾', '═', '¦', '║', '―', '¥', '▓', \n        '—', '‹', '─', '▒', '：', '⊕', '▼', '▪', '†', '■', '’', '▀', '¨', '▄', '♫', '☆', '¯', '♦', '¤', '▲', '¸', '⋅', '‘', '∞', \n        '∙', '）', '↓', '、', '│', '（', '»', '，', '♪', '╩', '╚', '・', '╦', '╣', '╔', '╗', '▬', '❤', '≤', '‡', '√', '◄', '━', \n        '⇒', '▶', '≥', '╝', '♡', '◊', '。', '✈', '≡', '☺', '✔', '↵', '≈', '✓', '♣', '☎', '℃', '◦', '└', '‟', '～', '！', '○', \n        '◆', '№', '♠', '▌', '✿', '▸', '⁄', '□', '❖', '✦', '．', '÷', '｜', '┃', '／', '￥', '╠', '↩', '✭', '▐', '☼', '☻', '┐', \n        '├', '«', '∼', '┌', '℉', '☮', '฿', '≦', '♬', '✧', '〉', '－', '⌂', '✖', '･', '◕', '※', '‖', '◀', '‰', '\\x97', '↺', \n        '∆', '┘', '┬', '╬', '،', '⌘', '⊂', '＞', '〈', '⎙', '？', '☠', '⇐', '▫', '∗', '∈', '≠', '♀', '♔', '˚', '℗', '┗', '＊', \n        '┼', '❀', '＆', '∩', '♂', '‿', '∑', '‣', '➜', '┛', '⇓', '☯', '⊖', '☀', '┳', '；', '∇', '⇑', '✰', '◇', '♯', '☞', '´', \n        '↔', '┏', '｡', '◘', '∂', '✌', '♭', '┣', '┴', '┓', '✨', '\\xa0', '˜', '❥', '┫', '℠', '✒', '［', '∫', '\\x93', '≧', '］', \n        '\\x94', '∀', '♛', '\\x96', '∨', '◎', '↻', '⇩', '＜', '≫', '✩', '✪', '♕', '؟', '₤', '☛', '╮', '␊', '＋', '┈', '％', \n        '╋', '▽', '⇨', '┻', '⊗', '￡', '।', '▂', '✯', '▇', '＿', '➤', '✞', '＝', '▷', '△', '◙', '▅', '✝', '∧', '␉', '☭', \n        '┊', '╯', '☾', '➔', '∴', '\\x92', '▃', '↳', '＾', '׳', '➢', '╭', '➡', '＠', '⊙', '☢', '˝', '∏', '„', '∥', '❝', '☐', \n        '▆', '╱', '⋙', '๏', '☁', '⇔', '▔', '\\x91', '➚', '◡', '╰', '\\x85', '♢', '˙', '۞', '✘', '✮', '☑', '⋆', 'ⓘ', '❒', \n        '☣', '✉', '⌊', '➠', '∣', '❑', '◢', 'ⓒ', '\\x80', '〒', '∕', '▮', '⦿', '✫', '✚', '⋯', '♩', '☂', '❞', '‗', '܂', '☜', \n        '‾', '✜', '╲', '∘', '⟩', '＼', '⟨', '·', '✗', '♚', '∅', 'ⓔ', '◣', '͡', '‛', '❦', '◠', '✄', '❄', '∃', '␣', '≪', '｢', \n        '≅', '◯', '☽', '∎', '｣', '❧', '̅', 'ⓐ', '↘', '⚓', '▣', '˘', '∪', '⇢', '✍', '⊥', '＃', '⎯', '↠', '۩', '☰', '◥', \n        '⊆', '✽', '⚡', '↪', '❁', '☹', '◼', '☃', '◤', '❏', 'ⓢ', '⊱', '➝', '̣', '✡', '∠', '｀', '▴', '┤', '∝', '♏', 'ⓐ', \n        '✎', ';', '␤', '＇', '❣', '✂', '✤', 'ⓞ', '☪', '✴', '⌒', '˛', '♒', '＄', '✶', '▻', 'ⓔ', '◌', '◈', '❚', '❂', '￦', \n        '◉', '╜', '̃', '✱', '╖', '❉', 'ⓡ', '↗', 'ⓣ', '♻', '➽', '׀', '✲', '✬', '☉', '▉', '≒', '☥', '⌐', '♨', '✕', 'ⓝ', \n        '⊰', '❘', '＂', '⇧', '̵', '➪', '▁', '▏', '⊃', 'ⓛ', '‚', '♰', '́', '✏', '⏑', '̶', 'ⓢ', '⩾', '￠', '❍', '≃', '⋰', '♋', \n        '､', '̂', '❋', '✳', 'ⓤ', '╤', '▕', '⌣', '✸', '℮', '⁺', '▨', '╨', 'ⓥ', '♈', '❃', '☝', '✻', '⊇', '≻', '♘', '♞', \n        '◂', '✟', '⌠', '✠', '☚', '✥', '❊', 'ⓒ', '⌈', '❅', 'ⓡ', '♧', 'ⓞ', '▭', '❱', 'ⓣ', '∟', '☕', '♺', '∵', '⍝', 'ⓑ', \n        '✵', '✣', '٭', '♆', 'ⓘ', '∶', '⚜', '◞', '்', '✹', '➥', '↕', '̳', '∷', '✋', '➧', '∋', '̿', 'ͧ', '┅', '⥤', '⬆', '⋱', \n        '☄', '↖', '⋮', '۔', '♌', 'ⓛ', '╕', '♓', '❯', '♍', '▋', '✺', '⭐', '✾', '♊', '➣', '▿', 'ⓑ', '♉', '⏠', '◾', '▹', \n        '⩽', '↦', '╥', '⍵', '⌋', '։', '➨', '∮', '⇥', 'ⓗ', 'ⓓ', '⁻', '⎝', '⌥', '⌉', '◔', '◑', '✼', '♎', '♐', '╪', '⊚', \n        '☒', '⇤', 'ⓜ', '⎠', '◐', '⚠', '╞', '◗', '⎕', 'ⓨ', '☟', 'ⓟ', '♟', '❈', '↬', 'ⓓ', '◻', '♮', '❙', '♤', '∉', '؛', \n        '⁂', 'ⓝ', '־', '♑', '╫', '╓', '╳', '⬅', '☔', '☸', '┄', '╧', '׃', '⎢', '❆', '⋄', '⚫', '̏', '☏', '➞', '͂', '␙', \n        'ⓤ', '◟', '̊', '⚐', '✙', '↙', '̾', '℘', '✷', '⍺', '❌', '⊢', '▵', '✅', 'ⓖ', '☨', '▰', '╡', 'ⓜ', '☤', '∽', '╘', \n        '˹', '↨', '♙', '⬇', '♱', '⌡', '⠀', '╛', '❕', '┉', 'ⓟ', '̀', '♖', 'ⓚ', '┆', '⎜', '◜', '⚾', '⤴', '✇', '╟', '⎛', \n        '☩', '➲', '➟', 'ⓥ', 'ⓗ', '⏝', '◃', '╢', '↯', '✆', '˃', '⍴', '❇', '⚽', '╒', '̸', '♜', '☓', '➳', '⇄', '☬', '⚑', \n        '✐', '⌃', '◅', '▢', '❐', '∊', '☈', '॥', '⎮', '▩', 'ு', '⊹', '‵', '␔', '☊', '➸', '̌', '☿', '⇉', '⊳', '╙', 'ⓦ', \n        '⇣', '｛', '̄', '↝', '⎟', '▍', '❗', '״', '΄', '▞', '◁', '⛄', '⇝', '⎪', '♁', '⇠', '☇', '✊', 'ி', '｝', '⭕', '➘', \n        '⁀', '☙', '❛', '❓', '⟲', '⇀', '≲', 'ⓕ', '⎥', '\\u06dd', 'ͤ', '₋', '̱', '̎', '♝', '≳', '▙', '➭', '܀', 'ⓖ', '⇛', '▊', \n        '⇗', '̷', '⇱', '℅', 'ⓧ', '⚛', '̐', '̕', '⇌', '␀', '≌', 'ⓦ', '⊤', '̓', '☦', 'ⓕ', '▜', '➙', 'ⓨ', '⌨', '◮', '☷', \n        '◍', 'ⓚ', '≔', '⏩', '⍳', '℞', '┋', '˻', '▚', '≺', 'ْ', '▟', '➻', '̪', '⏪', '̉', '⎞', '┇', '⍟', '⇪', '▎', '⇦', '␝', \n        '⤷', '≖', '⟶', '♗', '̴', '♄', 'ͨ', '̈', '❜', '̡', '▛', '✁', '➩', 'ா', '˂', '↥', '⏎', '⎷', '̲', '➖', '↲', '⩵', '̗', '❢', \n        '≎', '⚔', '⇇', '̑', '⊿', '̖', '☍', '➹', '⥊', '⁁', '✢']","metadata":{"execution":{"iopub.status.busy":"2022-01-08T14:51:32.340628Z","iopub.execute_input":"2022-01-08T14:51:32.341144Z","iopub.status.idle":"2022-01-08T14:51:32.470386Z","shell.execute_reply.started":"2022-01-08T14:51:32.341115Z","shell.execute_reply":"2022-01-08T14:51:32.469528Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def clean_punct(x):\n    emptyString=\"\"\n    for punct in puncts:\n        if punct in x:\n            x = x.replace(punct, emptyString)\n    return x","metadata":{"execution":{"iopub.status.busy":"2022-01-08T14:51:32.471667Z","iopub.execute_input":"2022-01-08T14:51:32.472133Z","iopub.status.idle":"2022-01-08T14:51:32.486826Z","shell.execute_reply.started":"2022-01-08T14:51:32.472088Z","shell.execute_reply":"2022-01-08T14:51:32.485951Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"nlp = spacy.load(\"en_core_web_sm\", disable=['parser','ner'])\nstopwords = set(stopwords.words('english'))\nlemmatizer = WordNetLemmatizer()\n#Không xóa stop word \ndef clean_text(text):        \n    # chuyển về dạng chữ thường\n    text = text.lower()       \n    #Clean tag (math,URL)\n    text = clean_tag(text)\n    # Chuyển các từ viết tắt trong từ điển về dạng thường\n    text = clean_contractions(text, contraction_mapping)\n    #Xóa dấu và ký tự đặc biệt\n    text = clean_punct(text)\n    tokens = word_tokenize(text)\n    # Bỏ stop word  \n    #tokens_not_sw = [word for word in tokens if not word in stopwords]  \n    \n    # chuyển từ số nhiều về dạng thường\n    #text = [lemmatizer.lemmatize(word) for word in tokens_not_sw ] \n    text = [lemmatizer.lemmatize(word) for word in tokens ] \n    text = \" \".join(text)\n\n    return text","metadata":{"execution":{"iopub.status.busy":"2022-01-08T14:51:32.488254Z","iopub.execute_input":"2022-01-08T14:51:32.488715Z","iopub.status.idle":"2022-01-08T14:51:33.532876Z","shell.execute_reply.started":"2022-01-08T14:51:32.488651Z","shell.execute_reply":"2022-01-08T14:51:33.531832Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Kiểm tra các các câu hỏi khi đã clear","metadata":{}},{"cell_type":"code","source":"question_sample = train_data.question_text.sample(1).values[0]\nquestion_sample","metadata":{"execution":{"iopub.status.busy":"2022-01-08T14:51:33.534342Z","iopub.execute_input":"2022-01-08T14:51:33.534784Z","iopub.status.idle":"2022-01-08T14:51:33.577998Z","shell.execute_reply.started":"2022-01-08T14:51:33.534744Z","shell.execute_reply":"2022-01-08T14:51:33.577178Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"clean_text(question_sample)","metadata":{"execution":{"iopub.status.busy":"2022-01-08T14:51:33.579483Z","iopub.execute_input":"2022-01-08T14:51:33.579801Z","iopub.status.idle":"2022-01-08T14:51:35.673172Z","shell.execute_reply.started":"2022-01-08T14:51:33.579773Z","shell.execute_reply":"2022-01-08T14:51:35.672011Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Làm sạch dữ liệu trên tập train và tập public test","metadata":{}},{"cell_type":"code","source":"train_data['clean_text'] = train_data['question_text'].apply(clean_text)\ntest_data['clean_text'] = test_data['question_text'].apply(clean_text)","metadata":{"execution":{"iopub.status.busy":"2022-01-08T14:51:35.674600Z","iopub.execute_input":"2022-01-08T14:51:35.675024Z","iopub.status.idle":"2022-01-08T14:59:41.166316Z","shell.execute_reply.started":"2022-01-08T14:51:35.674981Z","shell.execute_reply":"2022-01-08T14:59:41.165072Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"In ra một vài data của tập train khi đã làm sạch","metadata":{}},{"cell_type":"code","source":"train_data.head(5)","metadata":{"execution":{"iopub.status.busy":"2022-01-08T14:59:41.167674Z","iopub.execute_input":"2022-01-08T14:59:41.168021Z","iopub.status.idle":"2022-01-08T14:59:41.180710Z","shell.execute_reply.started":"2022-01-08T14:59:41.167991Z","shell.execute_reply":"2022-01-08T14:59:41.179842Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#train_text = train_data['question_text']\n#test_text = test_data['question_text']\ntrain_text = train_data['clean_text']\ntest_text = test_data['clean_text']\ntrain_target = train_data['target']\nall_text = train_text.append(test_text)","metadata":{"execution":{"iopub.status.busy":"2022-01-08T14:59:41.181826Z","iopub.execute_input":"2022-01-08T14:59:41.182092Z","iopub.status.idle":"2022-01-08T14:59:41.237084Z","shell.execute_reply.started":"2022-01-08T14:59:41.182067Z","shell.execute_reply":"2022-01-08T14:59:41.236199Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Huấn luyện mô hình với các model","metadata":{}},{"cell_type":"markdown","source":"### import các thư viện","metadata":{}},{"cell_type":"code","source":"from sklearn.feature_extraction.text import CountVectorizer\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.metrics import accuracy_score, confusion_matrix\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.svm import LinearSVC\nfrom sklearn.metrics import f1_score\nfrom sklearn.metrics import classification_report\nfrom sklearn.model_selection import KFold\nfrom sklearn.feature_extraction.text import TfidfVectorizer,CountVectorizer","metadata":{"execution":{"iopub.status.busy":"2022-01-08T14:59:41.238071Z","iopub.execute_input":"2022-01-08T14:59:41.238461Z","iopub.status.idle":"2022-01-08T14:59:41.245574Z","shell.execute_reply.started":"2022-01-08T14:59:41.238433Z","shell.execute_reply":"2022-01-08T14:59:41.244790Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Model SVC và CountVector","metadata":{}},{"cell_type":"markdown","source":"Sử dụng count vectorizer với ngram_range = (1-2). <br>\nSử dụng pipeline count vectorizer và SVC model","metadata":{}},{"cell_type":"code","source":"count_vectorizer = CountVectorizer(analyzer=\"word\", ngram_range=(1,2))\nSVC_model = LinearSVC(C=1, random_state=1)\nvector_svc_model = Pipeline([('count_vectorizer', count_vectorizer),('SVC', SVC_model)])","metadata":{"execution":{"iopub.status.busy":"2022-01-08T14:59:41.246630Z","iopub.execute_input":"2022-01-08T14:59:41.247064Z","iopub.status.idle":"2022-01-08T14:59:41.257021Z","shell.execute_reply.started":"2022-01-08T14:59:41.247022Z","shell.execute_reply":"2022-01-08T14:59:41.256121Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#X = train_data['question_text']\nX = train_data['clean_text']\ny = train_data['target']","metadata":{"execution":{"iopub.status.busy":"2022-01-08T14:59:41.258317Z","iopub.execute_input":"2022-01-08T14:59:41.259019Z","iopub.status.idle":"2022-01-08T14:59:41.272552Z","shell.execute_reply.started":"2022-01-08T14:59:41.258977Z","shell.execute_reply":"2022-01-08T14:59:41.271585Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Chia dữ liệu huấn luyện thành 2 phần: 80% để huấn luyện và 20% để đánh giá, tính f1_score","metadata":{}},{"cell_type":"code","source":"train_X, test_X, train_y, test_y = train_test_split(X, y, test_size=0.2)","metadata":{"execution":{"iopub.status.busy":"2022-01-08T14:59:41.273993Z","iopub.execute_input":"2022-01-08T14:59:41.274648Z","iopub.status.idle":"2022-01-08T14:59:41.655486Z","shell.execute_reply.started":"2022-01-08T14:59:41.274608Z","shell.execute_reply":"2022-01-08T14:59:41.654767Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"vector_svc_model.fit(train_X, train_y)","metadata":{"execution":{"iopub.status.busy":"2022-01-08T14:59:41.656774Z","iopub.execute_input":"2022-01-08T14:59:41.657147Z","iopub.status.idle":"2022-01-08T15:02:56.707299Z","shell.execute_reply.started":"2022-01-08T14:59:41.657108Z","shell.execute_reply":"2022-01-08T15:02:56.706371Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"svc_predictions = vector_svc_model.predict(test_X)\naccuracy_score(test_y, svc_predictions)\nf1_score(test_y, svc_predictions)","metadata":{"execution":{"iopub.status.busy":"2022-01-08T15:02:56.708482Z","iopub.execute_input":"2022-01-08T15:02:56.708772Z","iopub.status.idle":"2022-01-08T15:03:08.437196Z","shell.execute_reply.started":"2022-01-08T15:02:56.708744Z","shell.execute_reply":"2022-01-08T15:03:08.436198Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Logistic regression và count vectorizer","metadata":{}},{"cell_type":"markdown","source":"Sử dụng pipeline logistic regression với count vectorizer","metadata":{}},{"cell_type":"code","source":"logit_model = LogisticRegression(C=1, random_state=0)\nvector_logit_model = Pipeline([('count_vectorizer', count_vectorizer),('logit', logit_model)])","metadata":{"execution":{"iopub.status.busy":"2022-01-08T15:03:08.438941Z","iopub.execute_input":"2022-01-08T15:03:08.439226Z","iopub.status.idle":"2022-01-08T15:03:08.444597Z","shell.execute_reply.started":"2022-01-08T15:03:08.439199Z","shell.execute_reply":"2022-01-08T15:03:08.443652Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"vector_logit_model.fit(train_X, train_y)","metadata":{"execution":{"iopub.status.busy":"2022-01-08T15:03:08.445752Z","iopub.execute_input":"2022-01-08T15:03:08.446250Z","iopub.status.idle":"2022-01-08T15:05:50.894720Z","shell.execute_reply.started":"2022-01-08T15:03:08.446215Z","shell.execute_reply":"2022-01-08T15:05:50.893730Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"logit_predictions = vector_logit_model.predict(test_X)\naccuracy_score(test_y, logit_predictions)\nf1_score(test_y, logit_predictions)","metadata":{"execution":{"iopub.status.busy":"2022-01-08T15:05:50.896016Z","iopub.execute_input":"2022-01-08T15:05:50.896453Z","iopub.status.idle":"2022-01-08T15:06:01.660039Z","shell.execute_reply.started":"2022-01-08T15:05:50.896426Z","shell.execute_reply":"2022-01-08T15:06:01.659173Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"So sánh mô hình SVC và logistic regression ","metadata":{}},{"cell_type":"code","source":"print(\"Mo hinh SVC\")\nprint(classification_report(test_y, svc_predictions))\nprint( \"f1_score = \", f1_score(test_y, svc_predictions))\nprint(\"Mo hinh logistic regression\")\nprint(classification_report(test_y, logit_predictions))\nprint(\"f1_score = \", f1_score(test_y, logit_predictions))","metadata":{"execution":{"iopub.status.busy":"2022-01-08T15:06:01.661284Z","iopub.execute_input":"2022-01-08T15:06:01.661562Z","iopub.status.idle":"2022-01-08T15:06:02.464640Z","shell.execute_reply.started":"2022-01-08T15:06:01.661534Z","shell.execute_reply":"2022-01-08T15:06:02.463649Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Logistic regression vs Tfidf ","metadata":{}},{"cell_type":"markdown","source":"### TFIDF","metadata":{}},{"cell_type":"code","source":"tfidf_vectorizer = TfidfVectorizer(ngram_range=(1,2))\ntfidf_vectorizer.fit(all_text)\n\ncount_vectorizer.fit(all_text)\n\ntrain_text_features_cv = count_vectorizer.transform(train_text)\ntest_text_features_cv = count_vectorizer.transform(test_text)\n\ntrain_text_features_tf = tfidf_vectorizer.transform(train_text)\ntest_text_features_tf = tfidf_vectorizer.transform(test_text)","metadata":{"execution":{"iopub.status.busy":"2022-01-08T15:06:02.465856Z","iopub.execute_input":"2022-01-08T15:06:02.466119Z","iopub.status.idle":"2022-01-08T15:11:25.140823Z","shell.execute_reply.started":"2022-01-08T15:06:02.466085Z","shell.execute_reply":"2022-01-08T15:11:25.139577Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_text.head()","metadata":{"execution":{"iopub.status.busy":"2022-01-08T15:11:25.151832Z","iopub.execute_input":"2022-01-08T15:11:25.152173Z","iopub.status.idle":"2022-01-08T15:11:25.162112Z","shell.execute_reply.started":"2022-01-08T15:11:25.152143Z","shell.execute_reply":"2022-01-08T15:11:25.161028Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### K-Fold Cross Validation","metadata":{}},{"cell_type":"markdown","source":"Chia training data thành 5 fold, sử dụng id-idf và train với mô hình hồi quy logistic","metadata":{}},{"cell_type":"code","source":"kfold = KFold(n_splits = 10, shuffle = True, random_state = 1000)\ntest_preds = 0\noof_preds = np.zeros([train_data.shape[0],])\n\nfor i, (train_idx,valid_idx) in enumerate(kfold.split(train_data)):\n    x_train, x_valid = train_text_features_tf[train_idx,:], train_text_features_tf[valid_idx,:]\n    y_train, y_valid = train_target[train_idx], train_target[valid_idx]\n    logit_model = LogisticRegression()\n    print('fitting.......')\n    logit_model.fit(x_train,y_train)\n    print('predicting......')\n    print('\\n')\n    #lưu kết quả vào oof_preds\n    oof_preds[valid_idx] = logit_model.predict_proba(x_valid)[:,1]\n    test_preds += 0.2*logit_model.predict_proba(test_text_features_tf)[:,1]","metadata":{"execution":{"iopub.status.busy":"2022-01-08T15:21:54.281555Z","iopub.execute_input":"2022-01-08T15:21:54.281844Z","iopub.status.idle":"2022-01-08T15:43:29.458292Z","shell.execute_reply.started":"2022-01-08T15:21:54.281817Z","shell.execute_reply":"2022-01-08T15:43:29.457110Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Chọn threshold  có f1_score lớn nhất","metadata":{}},{"cell_type":"code","source":"for (i) in range(20,30, 1):\n    pred_train = (oof_preds > i / 100).astype(np.int)\n    print(\"threshold\", i/100, \"f1_score\", f1_score(train_target, pred_train))","metadata":{"execution":{"iopub.status.busy":"2022-01-08T15:43:57.579189Z","iopub.execute_input":"2022-01-08T15:43:57.579579Z","iopub.status.idle":"2022-01-08T15:44:01.784570Z","shell.execute_reply.started":"2022-01-08T15:43:57.579543Z","shell.execute_reply":"2022-01-08T15:44:01.783625Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"threshold = 0.22\npred_train = (oof_preds > .22).astype(np.int)\nf1_score(train_target, pred_train)","metadata":{"execution":{"iopub.status.busy":"2022-01-08T15:44:34.163075Z","iopub.execute_input":"2022-01-08T15:44:34.163416Z","iopub.status.idle":"2022-01-08T15:44:34.604571Z","shell.execute_reply.started":"2022-01-08T15:44:34.163387Z","shell.execute_reply":"2022-01-08T15:44:34.603465Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Submit","metadata":{}},{"cell_type":"code","source":"submission = pd.DataFrame.from_dict({'qid': test_data['qid']})\nsubmission['prediction'] = (test_preds>threshold).astype(np.int)\nsubmission.to_csv('submission.csv', index=False)\nsubmission['prediction'] = (test_preds>threshold)","metadata":{"execution":{"iopub.status.busy":"2022-01-08T15:44:36.537480Z","iopub.execute_input":"2022-01-08T15:44:36.538226Z","iopub.status.idle":"2022-01-08T15:44:37.455190Z","shell.execute_reply.started":"2022-01-08T15:44:36.538181Z","shell.execute_reply":"2022-01-08T15:44:37.454401Z"},"trusted":true},"execution_count":null,"outputs":[]}]}