{"metadata":{"kernelspec":{"name":"python3","display_name":"Python 3","language":"python"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":10737,"databundleVersionId":290346,"sourceType":"competition"}],"dockerImageVersionId":30698,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\nfrom sklearn.feature_extraction.text import TfidfVectorizer\nfrom sklearn.svm import LinearSVC\nfrom sklearn.metrics import f1_score\nfrom sklearn.model_selection import train_test_split\nfrom wordcloud import STOPWORDS\nimport seaborn as sns\nimport plotly.graph_objs as go\nimport plotly.offline as py\n\nimport re\nfrom string import digits\nimport os\nimport seaborn as sb\nfrom sklearn.cluster import KMeans\nfrom yellowbrick.cluster import KElbowVisualizer\nimport matplotlib.pyplot as plt\nfrom scipy.sparse import coo_matrix, hstack\nfrom sklearn import preprocessing\nfrom tqdm import tqdm","metadata":{"execution":{"iopub.status.busy":"2024-05-26T19:06:07.149428Z","iopub.execute_input":"2024-05-26T19:06:07.149932Z","iopub.status.idle":"2024-05-26T19:06:09.357997Z","shell.execute_reply.started":"2024-05-26T19:06:07.149891Z","shell.execute_reply":"2024-05-26T19:06:09.356772Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/quora-insincere-questions-classification/train.csv')\ntest = pd.read_csv('/kaggle/input/quora-insincere-questions-classification/test.csv')\ntest.head()","metadata":{"execution":{"iopub.status.busy":"2024-05-26T19:06:09.788018Z","iopub.execute_input":"2024-05-26T19:06:09.789223Z","iopub.status.idle":"2024-05-26T19:06:16.258501Z","shell.execute_reply.started":"2024-05-26T19:06:09.789182Z","shell.execute_reply":"2024-05-26T19:06:16.257077Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Tổng số dữ liệu trong tập train: \",train.shape[0])\nprint(\"Số câu hỏi bình thường: \", len(train[train.target == 0]))\nprint(\"Số câu hỏi toxic: \",len(train[train.target == 1]))\nprint(\"Tỉ lệ giữa 2 lớp: \",len(train[train.target == 1])/len(train[train.target == 0]))\nprint('\\n')\n\n#sb.countplot(train['target'])\ncnt_srs = train['target'].value_counts()\n\n## target distribution ##\nlabels = (np.array(cnt_srs.index))\nsizes = (np.array((cnt_srs / cnt_srs.sum())*100))\n\ntrace = go.Pie(labels=labels, values=sizes)\nlayout = go.Layout(\n    title='Target distribution',\n    font=dict(size=18),\n    width=350,\n    height=500,\n)\ndata = [trace]\nfig = go.Figure(data=data, layout=layout)\npy.iplot(fig, filename=\"usertype\")","metadata":{"execution":{"iopub.status.busy":"2024-05-26T19:07:24.736318Z","iopub.execute_input":"2024-05-26T19:07:24.736773Z","iopub.status.idle":"2024-05-26T19:07:25.047345Z","shell.execute_reply.started":"2024-05-26T19:07:24.736730Z","shell.execute_reply":"2024-05-26T19:07:25.046105Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Xứ lý kí tự toán học và link URL\n\ndef clean_tag(x):\n  if '[math]' in x:\n    x = re.sub('\\[math\\].*?math\\]', 'MATH EQUATION', x) #replacing with [MATH EQUATION]    \n  if 'http' in x or 'www' in x:\n    x = re.sub('(?:(?:https?|ftp):\\/\\/)?[\\w/\\-?=%.]+\\.[\\w/\\-?=%.]+', 'URL', x) #replacing with [url]\n  child_pattern = r'(arctan|arccos|arctg|arccot|arcsin|sin|cos|tan|cot|sqrt|log|exp|ln|diff|sec|frac)\\s*(\\{[^\\{\\}]*\\}|\\[[^\\[\\]]*\\]|\\([^()]*\\))'\n  x = re.sub(child_pattern, 'MATH EQUATION', x)\n  return x\n","metadata":{"execution":{"iopub.status.busy":"2024-05-26T19:07:28.286369Z","iopub.execute_input":"2024-05-26T19:07:28.287761Z","iopub.status.idle":"2024-05-26T19:07:28.295632Z","shell.execute_reply.started":"2024-05-26T19:07:28.287717Z","shell.execute_reply":"2024-05-26T19:07:28.294196Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Lọai bỏ số \ndef delete_number(strings) :\n    return ''.join([char for char in strings if char not in digits])","metadata":{"execution":{"iopub.status.busy":"2024-05-26T19:07:30.678985Z","iopub.execute_input":"2024-05-26T19:07:30.680908Z","iopub.status.idle":"2024-05-26T19:07:30.687418Z","shell.execute_reply.started":"2024-05-26T19:07:30.680851Z","shell.execute_reply":"2024-05-26T19:07:30.686181Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Loại bỏ kí tự đặc biệt\n\npuncts = [',', '.', '\"', ':', ')', '(', '-', '!', '?', '|', ';', \"'\", '$', '&', '/', '[', ']', '>', '%', '=', '#', '*', '+', '\\\\', \n        '•', '~', '@', '£', '·', '_', '{', '}', '©', '^', '®', '`', '<', '→', '°', '€', '™', '›', '♥', '←', '×', '§', '″', '′', \n        '█', '…', '“', '★', '”', '–', '●', '►', '−', '¢', '¬', '░', '¡', '¶', '↑', '±', '¿', '▾', '═', '¦', '║', '―', '¥', '▓', \n        '—', '‹', '─', '▒', '：', '⊕', '▼', '▪', '†', '■', '’', '▀', '¨', '▄', '♫', '☆', '¯', '♦', '¤', '▲', '¸', '⋅', '‘', '∞', \n        '∙', '）', '↓', '、', '│', '（', '»', '，', '♪', '╩', '╚', '・', '╦', '╣', '╔', '╗', '▬', '❤', '≤', '‡', '√', '◄', '━', \n        '⇒', '▶', '≥', '╝', '♡', '◊', '。', '✈', '≡', '☺', '✔', '↵', '≈', '✓', '♣', '☎', '℃', '◦', '└', '‟', '～', '！', '○', \n        '◆', '№', '♠', '▌', '✿', '▸', '⁄', '□', '❖', '✦', '．', '÷', '｜', '┃', '／', '￥', '╠', '↩', '✭', '▐', '☼', '☻', '┐', \n        '├', '«', '∼', '┌', '℉', '☮', '฿', '≦', '♬', '✧', '〉', '－', '⌂', '✖', '･', '◕', '※', '‖', '◀', '‰', '\\x97', '↺', \n        '∆', '┘', '┬', '╬', '،', '⌘', '⊂', '＞', '〈', '⎙', '？', '☠', '⇐', '▫', '∗', '∈', '≠', '♀', '♔', '˚', '℗', '┗', '＊', \n        '┼', '❀', '＆', '∩', '♂', '‿', '∑', '‣', '➜', '┛', '⇓', '☯', '⊖', '☀', '┳', '；', '∇', '⇑', '✰', '◇', '♯', '☞', '´', \n        '↔', '┏', '｡', '◘', '∂', '✌', '♭', '┣', '┴', '┓', '✨', '\\xa0', '˜', '❥', '┫', '℠', '✒', '［', '∫', '\\x93', '≧', '］', \n        '\\x94', '∀', '♛', '\\x96', '∨', '◎', '↻', '⇩', '＜', '≫', '✩', '✪', '♕', '؟', '₤', '☛', '╮', '␊', '＋', '┈', '％', \n        '╋', '▽', '⇨', '┻', '⊗', '￡', '।', '▂', '✯', '▇', '＿', '➤', '✞', '＝', '▷', '△', '◙', '▅', '✝', '∧', '␉', '☭', \n        '┊', '╯', '☾', '➔', '∴', '\\x92', '▃', '↳', '＾', '׳', '➢', '╭', '➡', '＠', '⊙', '☢', '˝', '∏', '„', '∥', '❝', '☐', \n        '▆', '╱', '⋙', '๏', '☁', '⇔', '▔', '\\x91', '➚', '◡', '╰', '\\x85', '♢', '˙', '۞', '✘', '✮', '☑', '⋆', 'ⓘ', '❒', \n        '☣', '✉', '⌊', '➠', '∣', '❑', '◢', 'ⓒ', '\\x80', '〒', '∕', '▮', '⦿', '✫', '✚', '⋯', '♩', '☂', '❞', '‗', '܂', '☜', \n        '‾', '✜', '╲', '∘', '⟩', '＼', '⟨', '·', '✗', '♚', '∅', 'ⓔ', '◣', '͡', '‛', '❦', '◠', '✄', '❄', '∃', '␣', '≪', '｢', \n        '≅', '◯', '☽', '∎', '｣', '❧', '̅', 'ⓐ', '↘', '⚓', '▣', '˘', '∪', '⇢', '✍', '⊥', '＃', '⎯', '↠', '۩', '☰', '◥', \n        '⊆', '✽', '⚡', '↪', '❁', '☹', '◼', '☃', '◤', '❏', 'ⓢ', '⊱', '➝', '̣', '✡', '∠', '｀', '▴', '┤', '∝', '♏', 'ⓐ', \n        '✎', ';', '␤', '＇', '❣', '✂', '✤', 'ⓞ', '☪', '✴', '⌒', '˛', '♒', '＄', '✶', '▻', 'ⓔ', '◌', '◈', '❚', '❂', '￦', \n        '◉', '╜', '̃', '✱', '╖', '❉', 'ⓡ', '↗', 'ⓣ', '♻', '➽', '׀', '✲', '✬', '☉', '▉', '≒', '☥', '⌐', '♨', '✕', 'ⓝ', \n        '⊰', '❘', '＂', '⇧', '̵', '➪', '▁', '▏', '⊃', 'ⓛ', '‚', '♰', '́', '✏', '⏑', '̶', 'ⓢ', '⩾', '￠', '❍', '≃', '⋰', '♋', \n        '､', '̂', '❋', '✳', 'ⓤ', '╤', '▕', '⌣', '✸', '℮', '⁺', '▨', '╨', 'ⓥ', '♈', '❃', '☝', '✻', '⊇', '≻', '♘', '♞', \n        '◂', '✟', '⌠', '✠', '☚', '✥', '❊', 'ⓒ', '⌈', '❅', 'ⓡ', '♧', 'ⓞ', '▭', '❱', 'ⓣ', '∟', '☕', '♺', '∵', '⍝', 'ⓑ', \n        '✵', '✣', '٭', '♆', 'ⓘ', '∶', '⚜', '◞', '்', '✹', '➥', '↕', '̳', '∷', '✋', '➧', '∋', '̿', 'ͧ', '┅', '⥤', '⬆', '⋱', \n        '☄', '↖', '⋮', '۔', '♌', 'ⓛ', '╕', '♓', '❯', '♍', '▋', '✺', '⭐', '✾', '♊', '➣', '▿', 'ⓑ', '♉', '⏠', '◾', '▹', \n        '⩽', '↦', '╥', '⍵', '⌋', '։', '➨', '∮', '⇥', 'ⓗ', 'ⓓ', '⁻', '⎝', '⌥', '⌉', '◔', '◑', '✼', '♎', '♐', '╪', '⊚', \n        '☒', '⇤', 'ⓜ', '⎠', '◐', '⚠', '╞', '◗', '⎕', 'ⓨ', '☟', 'ⓟ', '♟', '❈', '↬', 'ⓓ', '◻', '♮', '❙', '♤', '∉', '؛', \n        '⁂', 'ⓝ', '־', '♑', '╫', '╓', '╳', '⬅', '☔', '☸', '┄', '╧', '׃', '⎢', '❆', '⋄', '⚫', '̏', '☏', '➞', '͂', '␙', \n        'ⓤ', '◟', '̊', '⚐', '✙', '↙', '̾', '℘', '✷', '⍺', '❌', '⊢', '▵', '✅', 'ⓖ', '☨', '▰', '╡', 'ⓜ', '☤', '∽', '╘', \n        '˹', '↨', '♙', '⬇', '♱', '⌡', '⠀', '╛', '❕', '┉', 'ⓟ', '̀', '♖', 'ⓚ', '┆', '⎜', '◜', '⚾', '⤴', '✇', '╟', '⎛', \n        '☩', '➲', '➟', 'ⓥ', 'ⓗ', '⏝', '◃', '╢', '↯', '✆', '˃', '⍴', '❇', '⚽', '╒', '̸', '♜', '☓', '➳', '⇄', '☬', '⚑', \n        '✐', '⌃', '◅', '▢', '❐', '∊', '☈', '॥', '⎮', '▩', 'ு', '⊹', '‵', '␔', '☊', '➸', '̌', '☿', '⇉', '⊳', '╙', 'ⓦ', \n        '⇣', '｛', '̄', '↝', '⎟', '▍', '❗', '״', '΄', '▞', '◁', '⛄', '⇝', '⎪', '♁', '⇠', '☇', '✊', 'ி', '｝', '⭕', '➘', \n        '⁀', '☙', '❛', '❓', '⟲', '⇀', '≲', 'ⓕ', '⎥', '\\u06dd', 'ͤ', '₋', '̱', '̎', '♝', '≳', '▙', '➭', '܀', 'ⓖ', '⇛', '▊', \n        '⇗', '̷', '⇱', '℅', 'ⓧ', '⚛', '̐', '̕', '⇌', '␀', '≌', 'ⓦ', '⊤', '̓', '☦', 'ⓕ', '▜', '➙', 'ⓨ', '⌨', '◮', '☷', \n        '◍', 'ⓚ', '≔', '⏩', '⍳', '℞', '┋', '˻', '▚', '≺', 'ْ', '▟', '➻', '̪', '⏪', '̉', '⎞', '┇', '⍟', '⇪', '▎', '⇦', '␝', \n        '⤷', '≖', '⟶', '♗', '̴', '♄', 'ͨ', '̈', '❜', '̡', '▛', '✁', '➩', 'ா', '˂', '↥', '⏎', '⎷', '̲', '➖', '↲', '⩵', '̗', '❢', \n        '≎', '⚔', '⇇', '̑', '⊿', '̖', '☍', '➹', '⥊', '⁁', '✢', '^', 'ω', 'α', '¹' , '²', '³', 'µ', 'ª', '¼', '½', '¾', 'À', 'Á', \n        'Â', 'Ã', 'Ä', 'Å', 'Æ', 'Ç', 'È', 'É', 'Ê', 'Ë','Ì' ,'Í','Ï','Ð','Ñ','Ò','Ó','Ô','Õ','Ö','×','Ø','Ù','Ú','Ü','Ý','Þ',\n        'ß','à','á','â','ã','ä','å','æ','ç','è','é','ê','ë','ì','í','î','ï','ð' ,'ñ','ò' ,'ó','ô','õ' ,'ö' ,'ø' ,'ù' ,'ú','û' ,\n        'ü' ,'ý' ,'þ','ÿ']\n\n# def clean_punct(x):\n#     # Xây dựng một bảng dịch để loại bỏ các ký tự đặc biệt\n#     table = str.maketrans('', '', ''.join(puncts))\n\n#     # Loại bỏ các ký tự đặc biệt từ mỗi chuỗi trong danh sách\n#     new_strings = [s.translate(table) for s in x]\n#     return x\n\ndef clean_punct(x):\n    x = str(x)\n    for punct in puncts:\n        x = x.replace(punct, f' {punct} ')\n    return x\n","metadata":{"execution":{"iopub.status.busy":"2024-05-26T19:07:33.332949Z","iopub.execute_input":"2024-05-26T19:07:33.333765Z","iopub.status.idle":"2024-05-26T19:07:33.381783Z","shell.execute_reply.started":"2024-05-26T19:07:33.333729Z","shell.execute_reply":"2024-05-26T19:07:33.380280Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def find_unique_words():\n    unique_words = {'a'}\n    non_unique_words = {'a'}\n    for i in tqdm(range(0,len(train))):\n        for word in train.question_text.iloc[i].split():\n            if word in unique_words:\n                non_unique_words.add(word)\n            else:\n                unique_words.add(word)\n    a = pd.DataFrame(unique_words - non_unique_words)\n    return a\n\n#find_unique_words().head(30)","metadata":{"execution":{"iopub.status.busy":"2024-05-26T19:07:36.315094Z","iopub.execute_input":"2024-05-26T19:07:36.315508Z","iopub.status.idle":"2024-05-26T19:07:36.324488Z","shell.execute_reply.started":"2024-05-26T19:07:36.315477Z","shell.execute_reply":"2024-05-26T19:07:36.322523Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mispell_dict = {'colour': 'color', 'centre': 'center', 'favourite': 'favorite', 'travelling': 'traveling', 'counselling': 'counseling', 'theatre': 'theater', 'cancelled': 'canceled', 'labour': 'labor', 'organisation': 'organization', 'wwii': 'world war 2', 'citicise': 'criticize', 'youtu ': 'youtube ', 'Qoura': 'Quora', 'sallary': 'salary', 'Whta': 'What', 'narcisist': 'narcissist', 'howdo': 'how do', 'whatare': 'what are', 'howcan': 'how can', 'howmuch': 'how much', 'howmany': 'how many', 'whydo': 'why do', 'doI': 'do I', 'theBest': 'the best', 'howdoes': 'how does', 'mastrubation': 'masturbation', 'mastrubate': 'masturbate', \"mastrubating\": 'masturbating', 'pennis': 'penis', 'Etherium': 'bitcoin', 'narcissit': 'narcissist', 'bigdata': 'big data', '2k17': '2017', '2k18': '2018', 'qouta': 'quota', 'exboyfriend': 'ex boyfriend', 'airhostess': 'air hostess', \"whst\": 'what', 'watsapp': 'whatsapp', 'demonitisation': 'demonetization', 'demonitization': 'demonetization', 'demonetisation': 'demonetization', \n                'electroneum':'bitcoin','nanodegree':'degree','hotstar':'star','dream11':'dream','ftre':'fire','tensorflow':'framework','unocoin':'bitcoin',\n                'lnmiit':'limit','unacademy':'academy','altcoin':'bitcoin','altcoins':'bitcoin','litecoin':'bitcoin','coinbase':'bitcoin','cryptocurency':'cryptocurrency',\n                'simpliv':'simple','quoras':'quora','schizoids':'psychopath','remainers':'remainder','twinflame':'soulmate','quorans':'quora','brexit':'demonetized',\n                'iiest':'institute','dceu':'comics','pessat':'exam','uceed':'college','bhakts':'devotee','boruto':'anime',\n                'cryptocoin':'bitcoin','blockchains':'blockchain','fiancee':'fiance','redmi':'smartphone','oneplus':'smartphone','qoura':'quora','deepmind':'framework','ryzen':'cpu','whattsapp':'whatsapp',\n                'undertale':'adventure','zenfone':'smartphone','cryptocurencies':'cryptocurrencies','koinex':'bitcoin','zebpay':'bitcoin','binance':'bitcoin','whtsapp':'whatsapp',\n                'reactjs':'framework','bittrex':'bitcoin','bitconnect':'bitcoin','bitfinex':'bitcoin','yourquote':'your quote','whyis':'why is','jiophone':'smartphone',\n                'dogecoin':'bitcoin','onecoin':'bitcoin','poloniex':'bitcoin','7700k':'cpu','angular2':'framework','segwit2x':'bitcoin','hashflare':'bitcoin','940mx':'gpu',\n                'openai':'framework','hashflare':'bitcoin','1050ti':'gpu','nearbuy':'near buy','freebitco':'bitcoin','antminer':'bitcoin','filecoin':'bitcoin','whatapp':'whatsapp',\n                'empowr':'empower','1080ti':'gpu','crytocurrency':'cryptocurrency','8700k':'cpu','whatsaap':'whatsapp','g4560':'cpu','payymoney':'pay money',\n                'fuckboys':'fuck boys','intenship':'internship','zcash':'bitcoin','demonatisation':'demonetization','narcicist':'narcissist','mastuburation':'masturbation',\n                'trignometric':'trigonometric','cryptocurreny':'cryptocurrency','howdid':'how did','crytocurrencies':'cryptocurrencies','phycopath':'psychopath',\n                'bytecoin':'bitcoin','possesiveness':'possessiveness','scollege':'college','humanties':'humanities','altacoin':'bitcoin','demonitised':'demonetized',\n                'brasília':'brazilia','accolite':'accolyte','econimics':'economics','varrier':'warrier','quroa':'quora','statergy':'strategy','langague':'language',\n                'splatoon':'game','7600k':'cpu','gate2018':'gate 2018','in2018':'in 2018','narcassist':'narcissist','jiocoin':'bitcoin','hnlu':'hulu','7300hq':'cpu',\n                'weatern':'western','interledger':'blockchain','deplation':'deflation', 'cryptocurrencies':'cryptocurrency', 'bitcoin':'blockchain cryptocurrency',}\n\ndef correct_mispell(x):\n  words = x.split()\n  for i in range(0, len(words)):\n    if mispell_dict.get(words[i]) is not None:\n      words[i] = mispell_dict.get(words[i])\n    elif mispell_dict.get(words[i].lower()) is not None:\n      words[i] = mispell_dict.get(words[i].lower())\n        \n  words = \" \".join(words)\n  return words","metadata":{"execution":{"iopub.status.busy":"2024-05-26T19:07:41.810790Z","iopub.execute_input":"2024-05-26T19:07:41.812084Z","iopub.status.idle":"2024-05-26T19:07:41.833820Z","shell.execute_reply.started":"2024-05-26T19:07:41.812040Z","shell.execute_reply":"2024-05-26T19:07:41.832925Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"contraction_mapping = {\"We'd\": \"We had\", \"That'd\": \"That had\", \"AREN'T\": \"Are not\", \"HADN'T\": \"Had not\", \"Could've\": \"Could have\", \"LeT's\": \"Let us\", \"How'll\": \"How will\", \"They'll\": \"They will\", \"DOESN'T\": \"Does not\", \"HE'S\": \"He has\", \"O'Clock\": \"Of the clock\", \"Who'll\": \"Who will\", \"What'S\": \"What is\", \"Ain't\": \"Am not\", \"WEREN'T\": \"Were not\", \"Y'all\": \"You all\", \"Y'ALL\": \"You all\", \"Here's\": \"Here is\", \"It'd\": \"It had\", \"Should've\": \"Should have\", \"I'M\": \"I am\", \"ISN'T\": \"Is not\", \"Would've\": \"Would have\", \"He'll\": \"He will\", \"DON'T\": \"Do not\", \"She'd\": \"She had\", \"WOULDN'T\": \"Would not\", \"She'll\": \"She will\", \"IT's\": \"It is\", \"There'd\": \"There had\", \"It'll\": \"It will\", \"You'll\": \"You will\", \"He'd\": \"He had\", \"What'll\": \"What will\", \"Ma'am\": \"Madam\", \"CAN'T\": \"Can not\", \"THAT'S\": \"That is\", \"You've\": \"You have\", \"She's\": \"She is\", \"Weren't\": \"Were not\", \"They've\": \"They have\", \"Couldn't\": \"Could not\", \"When's\": \"When is\", \"Haven't\": \"Have not\", \"We'll\": \"We will\", \"That's\": \"That is\", \"We're\": \"We are\", \"They're\": \"They' are\", \"You'd\": \"You would\", \"How'd\": \"How did\", \"What're\": \"What are\", \"Hasn't\": \"Has not\", \"Wasn't\": \"Was not\", \"Won't\": \"Will not\", \"There's\": \"There is\", \"Didn't\": \"Did not\", \"Doesn't\": \"Does not\", \"You're\": \"You are\", \"He's\": \"He is\", \"SO's\": \"So is\", \"We've\": \"We have\", \"Who's\": \"Who is\", \"Wouldn't\": \"Would not\", \"Why's\": \"Why is\", \"WHO's\": \"Who is\", \"Let's\": \"Let us\", \"How's\": \"How is\", \"Can't\": \"Can not\", \"Where's\": \"Where is\", \"They'd\": \"They had\", \"Don't\": \"Do not\", \"Shouldn't\":\"Should not\", \"Aren't\":\"Are not\", \"ain't\": \"is not\", \"What's\": \"What is\", \"It's\": \"It is\", \"Isn't\":\"Is not\", \"aren't\": \"are not\",\"can't\": \"cannot\", \"'cause\": \"because\", \"could've\": \"could have\", \"couldn't\": \"could not\", \"didn't\": \"did not\",  \"doesn't\": \"does not\", \"don't\": \"do not\", \"hadn't\": \"had not\", \"hasn't\": \"has not\", \"haven't\": \"have not\", \"he'd\": \"he would\",\"he'll\": \"he will\", \"he's\": \"he is\", \"how'd\": \"how did\", \"how'd'y\": \"how do you\", \"how'll\": \"how will\", \"how's\": \"how is\",  \"I'd\": \"I would\", \"I'd've\": \"I would have\", \"I'll\": \"I will\", \"I'll've\": \"I will have\",\"I'm\": \"I am\", \"I've\": \"I have\", \"i'd\": \"i would\", \"i'd've\": \"i would have\", \"i'll\": \"i will\",  \"i'll've\": \"i will have\",\"i'm\": \"i am\", \"i've\": \"i have\", \"isn't\": \"is not\", \"it'd\": \"it would\", \"it'd've\": \"it would have\", \"it'll\": \"it will\", \"it'll've\": \"it will have\",\"it's\": \"it is\", \"let's\": \"let us\", \"ma'am\": \"madam\", \"mayn't\": \"may not\", \"might've\": \"might have\",\"mightn't\": \"might not\",\"mightn't've\": \"might not have\", \"must've\": \"must have\", \"mustn't\": \"must not\", \"mustn't've\": \"must not have\", \"needn't\": \"need not\", \"needn't've\": \"need not have\",\"o'clock\": \"of the clock\", \"oughtn't\": \"ought not\", \"oughtn't've\": \"ought not have\", \"shan't\": \"shall not\", \"sha'n't\": \"shall not\", \"shan't've\": \"shall not have\", \"she'd\": \"she would\", \"she'd've\": \"she would have\", \"she'll\": \"she will\", \"she'll've\": \"she will have\", \"she's\": \"she is\", \"should've\": \"should have\", \"shouldn't\": \"should not\", \"shouldn't've\": \"should not have\", \"so've\": \"so have\",\"so's\": \"so as\", \"this's\": \"this is\",\"that'd\": \"that would\", \"that'd've\": \"that would have\", \"that's\": \"that is\", \"there'd\": \"there would\", \"there'd've\": \"there would have\", \"there's\": \"there is\", \"here's\": \"here is\",\"they'd\": \"they would\", \"they'd've\": \"they would have\", \"they'll\": \"they will\", \"they'll've\": \"they will have\", \"they're\": \"they are\", \"they've\": \"they have\", \"to've\": \"to have\", \"wasn't\": \"was not\", \"we'd\": \"we would\", \"we'd've\": \"we would have\", \"we'll\": \"we will\", \"we'll've\": \"we will have\", \"we're\": \"we are\", \"we've\": \"we have\", \"weren't\": \"were not\", \"what'll\": \"what will\", \"what'll've\": \"what will have\", \"what're\": \"what are\",  \"what's\": \"what is\", \"what've\": \"what have\", \"when's\": \"when is\", \"when've\": \"when have\", \"where'd\": \"where did\", \"where's\": \"where is\", \"where've\": \"where have\", \"who'll\": \"who will\", \"who'll've\": \"who will have\", \"who's\": \"who is\", \"who've\": \"who have\", \"why's\": \"why is\", \"why've\": \"why have\", \"will've\": \"will have\", \"won't\": \"will not\", \"won't've\": \"will not have\", \"would've\": \"would have\", \"wouldn't\": \"would not\", \"wouldn't've\": \"would not have\", \"y'all\": \"you all\", \"y'all'd\": \"you all would\",\"y'all'd've\": \"you all would have\",\"y'all're\": \"you all are\",\"y'all've\": \"you all have\",\"you'd\": \"you would\", \"you'd've\": \"you would have\", \"you'll\": \"you will\", \"you'll've\": \"you will have\", \"you're\": \"you are\", \"you've\": \"you have\" }\n\ndef clean_contractions(text):\n    specials = [\"’\", \"‘\", \"´\", \"`\"]\n    for s in specials:\n        text = text.replace(s, \"'\")\n    \n    text = ' '.join([contraction_mapping[t] if t in contraction_mapping else t for t in text.split(\" \")])\n    return text","metadata":{"execution":{"iopub.status.busy":"2024-05-26T19:07:44.625860Z","iopub.execute_input":"2024-05-26T19:07:44.626261Z","iopub.status.idle":"2024-05-26T19:07:44.651587Z","shell.execute_reply.started":"2024-05-26T19:07:44.626233Z","shell.execute_reply":"2024-05-26T19:07:44.650212Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Xóa stopword\nstopwords = STOPWORDS  - {'ought', 'whom',\"wouldn't\", \"you'll\", \"you've\"}# bỏ đi một số từ khiến hiệu quả dự đoán giảm\n\ndef remove_stopwords(x):\n  x = [word.lower() for word in x.split() if word not in stopwords]\n  x = ' '.join(x)\n  return x\n","metadata":{"execution":{"iopub.status.busy":"2024-05-26T19:07:47.301265Z","iopub.execute_input":"2024-05-26T19:07:47.301952Z","iopub.status.idle":"2024-05-26T19:07:47.309641Z","shell.execute_reply.started":"2024-05-26T19:07:47.301920Z","shell.execute_reply":"2024-05-26T19:07:47.308028Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Xóa những từ trong chuỗi có 1 kí tự điều này để loại bỏ các biến x,t trong biểu thức toán\ndef remove_single_character_words(text):\n    if isinstance(text, str):  # Kiểm tra xem text có phải là một chuỗi không\n        # Tách chuỗi thành danh sách các từ và giữ lại những từ có độ dài lớn hơn 1\n        text = ' '.join([word for word in text.split() if len(word) > 1])\n    return text","metadata":{"execution":{"iopub.status.busy":"2024-05-26T19:07:49.827112Z","iopub.execute_input":"2024-05-26T19:07:49.827514Z","iopub.status.idle":"2024-05-26T19:07:49.834191Z","shell.execute_reply.started":"2024-05-26T19:07:49.827482Z","shell.execute_reply":"2024-05-26T19:07:49.832902Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Gọi tất cả các hàm tiền xử lý dữ liệu trên\ndef data_cleaning(x):\n  x = x.lower()\n  x = clean_tag(x)\n  x = delete_number(x)\n  x = correct_mispell(x)\n  x = clean_contractions(x)\n  x = clean_punct(x)\n  x = remove_single_character_words(x)\n#   x = lemma_text(x)\n  return x","metadata":{"execution":{"iopub.status.busy":"2024-05-26T19:07:52.387463Z","iopub.execute_input":"2024-05-26T19:07:52.387860Z","iopub.status.idle":"2024-05-26T19:07:52.394856Z","shell.execute_reply.started":"2024-05-26T19:07:52.387828Z","shell.execute_reply":"2024-05-26T19:07:52.393208Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Tạo ra 1 tập dữ liệu đã được cleaning\ntrain['question_text_cleaned'] = train['question_text'].apply(lambda x: data_cleaning(x))\n#print(*train['question_text_cleaned'], sep ='\\n')\ntest['question_text_cleaned'] = test['question_text'].apply(lambda x: data_cleaning(x))\ndisplay(train, test)","metadata":{"execution":{"iopub.status.busy":"2024-05-26T19:07:55.826353Z","iopub.execute_input":"2024-05-26T19:07:55.826765Z","iopub.status.idle":"2024-05-26T19:16:41.226557Z","shell.execute_reply.started":"2024-05-26T19:07:55.826734Z","shell.execute_reply":"2024-05-26T19:16:41.225440Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def generate_feature():\n    train['qlen'] = train['question_text'].str.len() \n    train['n_words'] = train['question_text'].apply(lambda row: len(row.split(\" \")))\n    train['numeric_words'] = train['question_text'].apply(lambda row: sum(c.isdigit() for c in row))\n    train['sp_char_words'] = train['question_text'].str.findall(r'[^a-zA-Z0-9 ]').str.len()\n    train['char_words'] = train['question_text'].apply(lambda row: len(str(row)))\n    train['unique_words'] = train['question_text'].apply(lambda row: len(set(str(row).split())))\n    train['stopwords'] = train['question_text'].apply(lambda x: len([c for c in str(x).lower().split() if c in STOPWORDS]))\n    test['qlen'] = test['question_text'].str.len() \n    test['n_words'] = test['question_text'].apply(lambda row: len(row.split(\" \")))\n    test['numeric_words'] = test['question_text'].apply(lambda row: sum(c.isdigit() for c in row))\n    test['sp_char_words'] = test['question_text'].str.findall(r'[^a-zA-Z0-9 ]').str.len()\n    test['char_words'] = test['question_text'].apply(lambda row: len(str(row)))\n    test['unique_words'] = test['question_text'].apply(lambda row: len(set(str(row).split())))\n    test['stopwords'] = test['question_text'].apply(lambda x: len([c for c in str(x).lower().split() if c in STOPWORDS])-1)\n    \ngenerate_feature()\ndisplay(train.head(),test.head())","metadata":{"execution":{"iopub.status.busy":"2024-05-26T19:16:41.228899Z","iopub.execute_input":"2024-05-26T19:16:41.229950Z","iopub.status.idle":"2024-05-26T19:17:20.535543Z","shell.execute_reply.started":"2024-05-26T19:16:41.229907Z","shell.execute_reply":"2024-05-26T19:17:20.534095Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Hàm dự đoán sử dụng linear\ntfidf = TfidfVectorizer(ngram_range=(1, 3))\ndef predict_linearSVC(X_train,y_train,X_test):\n    tfidf.fit(X_train)\n    X_train = tfidf.transform(X_train)\n    X_test = tfidf.transform(X_test)\n    svm = LinearSVC()\n    svm.fit(X_train,y_train)\n    return svm.predict(X_test)\n \n#Dự đoán trên tập validation\ndef validate_base_model():\n    train = pd.read_csv('/kaggle/input/quora-insincere-questions-classification/train.csv')\n    train = train.dropna(subset=['question_text'])\n    X = train.question_text\n    y = train.target\n    X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)\n    predict = predict_linearSVC(X_train,y_train,X_test)\n    return f1_score(predict,y_test)\n\n#print('F1-Score của base-model trên tập validation: ',validate_base_model())\n\ndef create_file_submission(predict):\n    submission = pd.DataFrame(test['qid'])\n    submission['prediction'] = predict\n    submission.to_csv('submission.csv', index=False)\n    \ndef submit_base_model():\n    X_train = train['question_text']\n    y_train = train.target\n    X_test = test['question_text']\n    predict = predict_linearSVC(X_train,y_train,X_test)\n    create_file_submission(predict)\n    \ndef validate_with_new_feature():\n    X = train[['question_text_cleaned','qlen','n_words','numeric_words','sp_char_words','char_words','unique_words','stopwords']]\n    y = train.target\n    X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)\n    tfidf.fit(X_train.question_text_cleaned)\n\n    A_train = tfidf.transform(X_train.question_text_cleaned)\n    A_test = tfidf.transform(X_test.question_text_cleaned)\n\n    scaler = preprocessing.MinMaxScaler()\n    print(scaler)\n    scaled_train = scaler.fit_transform(X_train[['qlen','n_words','numeric_words','sp_char_words','unique_words','stopwords']])\n    scaled_test = scaler.fit_transform(X_test[['qlen','n_words','numeric_words','sp_char_words','unique_words','stopwords']])\n\n    X_train = hstack([A_train,coo_matrix(scaled_train)])\n    X_test = hstack([A_test,coo_matrix(scaled_test)])\n    svm = LinearSVC()\n    svm.fit(X_train,y_train)\n    return f1_score(svm.predict(X_test),y_test)\n\ndef submit_with_new_feature():\n\n    tfidf.fit(train.question_text_cleaned)\n\n    A_train = tfidf.transform(train.question_text_cleaned)\n    A_test = tfidf.transform(test.question_text_cleaned)\n\n    scaler = preprocessing.MinMaxScaler()\n    print(scaler)\n    scaled_train = scaler.fit_transform(train[['qlen','n_words','numeric_words','sp_char_words','unique_words','stopwords']])\n    scaled_test = scaler.fit_transform(test[['qlen','n_words','numeric_words','sp_char_words','unique_words','stopwords']])\n\n    X_train = hstack([A_train,coo_matrix(scaled_train)])\n    X_test = hstack([A_test,coo_matrix(scaled_test)])\n    svm = LinearSVC()\n    svm.fit(X_train,train.target)\n    predict = svm.predict(X_test)\n\n    create_file_submission(predict)\n\nsubmit_with_new_feature()","metadata":{"execution":{"iopub.status.busy":"2024-05-26T19:18:49.795899Z","iopub.execute_input":"2024-05-26T19:18:49.796270Z","iopub.status.idle":"2024-05-26T19:23:55.783965Z","shell.execute_reply.started":"2024-05-26T19:18:49.796242Z","shell.execute_reply":"2024-05-26T19:23:55.782526Z"},"trusted":true},"execution_count":null,"outputs":[]}]}