{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\nfor dirname, _, filenames in os.walk('/kaggle/output'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-01-03T12:15:35.813498Z","iopub.execute_input":"2022-01-03T12:15:35.814256Z","iopub.status.idle":"2022-01-03T12:15:35.847534Z","shell.execute_reply.started":"2022-01-03T12:15:35.814147Z","shell.execute_reply":"2022-01-03T12:15:35.846481Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.feature_extraction.text import TfidfVectorizer\nfrom sklearn.svm import LinearSVC\nfrom sklearn.metrics import f1_score\nfrom sklearn.model_selection import train_test_split\nfrom wordcloud import STOPWORDS\nimport seaborn as sns\n\nimport re\nimport string\nimport os\nfrom sklearn.cluster import KMeans\nfrom yellowbrick.cluster import KElbowVisualizer\nimport matplotlib.pyplot as plt\nfrom scipy.sparse import coo_matrix, hstack\nfrom sklearn import preprocessing\nfrom tqdm import tqdm","metadata":{"execution":{"iopub.status.busy":"2022-01-03T12:15:35.849583Z","iopub.execute_input":"2022-01-03T12:15:35.850166Z","iopub.status.idle":"2022-01-03T12:15:37.315033Z","shell.execute_reply.started":"2022-01-03T12:15:35.850123Z","shell.execute_reply":"2022-01-03T12:15:37.314008Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#read train file and test file\ntrain = pd.read_csv('/kaggle/input/quora-insincere-questions-classification/train.csv')\ntest = pd.read_csv('/kaggle/input/quora-insincere-questions-classification/test.csv')","metadata":{"execution":{"iopub.status.busy":"2022-01-03T12:15:37.316238Z","iopub.execute_input":"2022-01-03T12:15:37.316457Z","iopub.status.idle":"2022-01-03T12:15:44.124183Z","shell.execute_reply.started":"2022-01-03T12:15:37.316432Z","shell.execute_reply":"2022-01-03T12:15:44.123366Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#use tfidf for encode question text\ntfidf = TfidfVectorizer(ngram_range=(1, 3))","metadata":{"execution":{"iopub.status.busy":"2022-01-03T12:15:44.125912Z","iopub.execute_input":"2022-01-03T12:15:44.126972Z","iopub.status.idle":"2022-01-03T12:15:44.132222Z","shell.execute_reply.started":"2022-01-03T12:15:44.126926Z","shell.execute_reply":"2022-01-03T12:15:44.131238Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Process math equation and url \n\ndef clean_tag(x):\n  if '[math]' in x:\n    x = re.sub('\\[math\\].*?math\\]', 'MATH EQUATION', x) #replacing with [MATH EQUATION]    \n  if 'http' in x or 'www' in x:\n    x = re.sub('(?:(?:https?|ftp):\\/\\/)?[\\w/\\-?=%.]+\\.[\\w/\\-?=%.]+', 'URL', x) #replacing with [url]\n  return x","metadata":{"execution":{"iopub.status.busy":"2022-01-03T12:15:44.133484Z","iopub.execute_input":"2022-01-03T12:15:44.133826Z","iopub.status.idle":"2022-01-03T12:15:44.144971Z","shell.execute_reply.started":"2022-01-03T12:15:44.133797Z","shell.execute_reply":"2022-01-03T12:15:44.143918Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Remove special character\n# Ref: https://www.kaggle.com/canming/ensemble-mean-iii-64-36\n\npuncts = [',', '.', '\"', ':', ')', '(', '-', '!', '?', '|', ';', \"'\", '$', '&', '/', '[', ']', '>', '%', '=', '#', '*', '+', '\\\\', \n        '•', '~', '@', '£', '·', '_', '{', '}', '©', '^', '®', '`', '<', '→', '°', '€', '™', '›', '♥', '←', '×', '§', '″', '′', \n        '█', '…', '“', '★', '”', '–', '●', '►', '−', '¢', '¬', '░', '¡', '¶', '↑', '±', '¿', '▾', '═', '¦', '║', '―', '¥', '▓', \n        '—', '‹', '─', '▒', '：', '⊕', '▼', '▪', '†', '■', '’', '▀', '¨', '▄', '♫', '☆', '¯', '♦', '¤', '▲', '¸', '⋅', '‘', '∞', \n        '∙', '）', '↓', '、', '│', '（', '»', '，', '♪', '╩', '╚', '・', '╦', '╣', '╔', '╗', '▬', '❤', '≤', '‡', '√', '◄', '━', \n        '⇒', '▶', '≥', '╝', '♡', '◊', '。', '✈', '≡', '☺', '✔', '↵', '≈', '✓', '♣', '☎', '℃', '◦', '└', '‟', '～', '！', '○', \n        '◆', '№', '♠', '▌', '✿', '▸', '⁄', '□', '❖', '✦', '．', '÷', '｜', '┃', '／', '￥', '╠', '↩', '✭', '▐', '☼', '☻', '┐', \n        '├', '«', '∼', '┌', '℉', '☮', '฿', '≦', '♬', '✧', '〉', '－', '⌂', '✖', '･', '◕', '※', '‖', '◀', '‰', '\\x97', '↺', \n        '∆', '┘', '┬', '╬', '،', '⌘', '⊂', '＞', '〈', '⎙', '？', '☠', '⇐', '▫', '∗', '∈', '≠', '♀', '♔', '˚', '℗', '┗', '＊', \n        '┼', '❀', '＆', '∩', '♂', '‿', '∑', '‣', '➜', '┛', '⇓', '☯', '⊖', '☀', '┳', '；', '∇', '⇑', '✰', '◇', '♯', '☞', '´', \n        '↔', '┏', '｡', '◘', '∂', '✌', '♭', '┣', '┴', '┓', '✨', '\\xa0', '˜', '❥', '┫', '℠', '✒', '［', '∫', '\\x93', '≧', '］', \n        '\\x94', '∀', '♛', '\\x96', '∨', '◎', '↻', '⇩', '＜', '≫', '✩', '✪', '♕', '؟', '₤', '☛', '╮', '␊', '＋', '┈', '％', \n        '╋', '▽', '⇨', '┻', '⊗', '￡', '।', '▂', '✯', '▇', '＿', '➤', '✞', '＝', '▷', '△', '◙', '▅', '✝', '∧', '␉', '☭', \n        '┊', '╯', '☾', '➔', '∴', '\\x92', '▃', '↳', '＾', '׳', '➢', '╭', '➡', '＠', '⊙', '☢', '˝', '∏', '„', '∥', '❝', '☐', \n        '▆', '╱', '⋙', '๏', '☁', '⇔', '▔', '\\x91', '➚', '◡', '╰', '\\x85', '♢', '˙', '۞', '✘', '✮', '☑', '⋆', 'ⓘ', '❒', \n        '☣', '✉', '⌊', '➠', '∣', '❑', '◢', 'ⓒ', '\\x80', '〒', '∕', '▮', '⦿', '✫', '✚', '⋯', '♩', '☂', '❞', '‗', '܂', '☜', \n        '‾', '✜', '╲', '∘', '⟩', '＼', '⟨', '·', '✗', '♚', '∅', 'ⓔ', '◣', '͡', '‛', '❦', '◠', '✄', '❄', '∃', '␣', '≪', '｢', \n        '≅', '◯', '☽', '∎', '｣', '❧', '̅', 'ⓐ', '↘', '⚓', '▣', '˘', '∪', '⇢', '✍', '⊥', '＃', '⎯', '↠', '۩', '☰', '◥', \n        '⊆', '✽', '⚡', '↪', '❁', '☹', '◼', '☃', '◤', '❏', 'ⓢ', '⊱', '➝', '̣', '✡', '∠', '｀', '▴', '┤', '∝', '♏', 'ⓐ', \n        '✎', ';', '␤', '＇', '❣', '✂', '✤', 'ⓞ', '☪', '✴', '⌒', '˛', '♒', '＄', '✶', '▻', 'ⓔ', '◌', '◈', '❚', '❂', '￦', \n        '◉', '╜', '̃', '✱', '╖', '❉', 'ⓡ', '↗', 'ⓣ', '♻', '➽', '׀', '✲', '✬', '☉', '▉', '≒', '☥', '⌐', '♨', '✕', 'ⓝ', \n        '⊰', '❘', '＂', '⇧', '̵', '➪', '▁', '▏', '⊃', 'ⓛ', '‚', '♰', '́', '✏', '⏑', '̶', 'ⓢ', '⩾', '￠', '❍', '≃', '⋰', '♋', \n        '､', '̂', '❋', '✳', 'ⓤ', '╤', '▕', '⌣', '✸', '℮', '⁺', '▨', '╨', 'ⓥ', '♈', '❃', '☝', '✻', '⊇', '≻', '♘', '♞', \n        '◂', '✟', '⌠', '✠', '☚', '✥', '❊', 'ⓒ', '⌈', '❅', 'ⓡ', '♧', 'ⓞ', '▭', '❱', 'ⓣ', '∟', '☕', '♺', '∵', '⍝', 'ⓑ', \n        '✵', '✣', '٭', '♆', 'ⓘ', '∶', '⚜', '◞', '்', '✹', '➥', '↕', '̳', '∷', '✋', '➧', '∋', '̿', 'ͧ', '┅', '⥤', '⬆', '⋱', \n        '☄', '↖', '⋮', '۔', '♌', 'ⓛ', '╕', '♓', '❯', '♍', '▋', '✺', '⭐', '✾', '♊', '➣', '▿', 'ⓑ', '♉', '⏠', '◾', '▹', \n        '⩽', '↦', '╥', '⍵', '⌋', '։', '➨', '∮', '⇥', 'ⓗ', 'ⓓ', '⁻', '⎝', '⌥', '⌉', '◔', '◑', '✼', '♎', '♐', '╪', '⊚', \n        '☒', '⇤', 'ⓜ', '⎠', '◐', '⚠', '╞', '◗', '⎕', 'ⓨ', '☟', 'ⓟ', '♟', '❈', '↬', 'ⓓ', '◻', '♮', '❙', '♤', '∉', '؛', \n        '⁂', 'ⓝ', '־', '♑', '╫', '╓', '╳', '⬅', '☔', '☸', '┄', '╧', '׃', '⎢', '❆', '⋄', '⚫', '̏', '☏', '➞', '͂', '␙', \n        'ⓤ', '◟', '̊', '⚐', '✙', '↙', '̾', '℘', '✷', '⍺', '❌', '⊢', '▵', '✅', 'ⓖ', '☨', '▰', '╡', 'ⓜ', '☤', '∽', '╘', \n        '˹', '↨', '♙', '⬇', '♱', '⌡', '⠀', '╛', '❕', '┉', 'ⓟ', '̀', '♖', 'ⓚ', '┆', '⎜', '◜', '⚾', '⤴', '✇', '╟', '⎛', \n        '☩', '➲', '➟', 'ⓥ', 'ⓗ', '⏝', '◃', '╢', '↯', '✆', '˃', '⍴', '❇', '⚽', '╒', '̸', '♜', '☓', '➳', '⇄', '☬', '⚑', \n        '✐', '⌃', '◅', '▢', '❐', '∊', '☈', '॥', '⎮', '▩', 'ு', '⊹', '‵', '␔', '☊', '➸', '̌', '☿', '⇉', '⊳', '╙', 'ⓦ', \n        '⇣', '｛', '̄', '↝', '⎟', '▍', '❗', '״', '΄', '▞', '◁', '⛄', '⇝', '⎪', '♁', '⇠', '☇', '✊', 'ி', '｝', '⭕', '➘', \n        '⁀', '☙', '❛', '❓', '⟲', '⇀', '≲', 'ⓕ', '⎥', '\\u06dd', 'ͤ', '₋', '̱', '̎', '♝', '≳', '▙', '➭', '܀', 'ⓖ', '⇛', '▊', \n        '⇗', '̷', '⇱', '℅', 'ⓧ', '⚛', '̐', '̕', '⇌', '␀', '≌', 'ⓦ', '⊤', '̓', '☦', 'ⓕ', '▜', '➙', 'ⓨ', '⌨', '◮', '☷', \n        '◍', 'ⓚ', '≔', '⏩', '⍳', '℞', '┋', '˻', '▚', '≺', 'ْ', '▟', '➻', '̪', '⏪', '̉', '⎞', '┇', '⍟', '⇪', '▎', '⇦', '␝', \n        '⤷', '≖', '⟶', '♗', '̴', '♄', 'ͨ', '̈', '❜', '̡', '▛', '✁', '➩', 'ா', '˂', '↥', '⏎', '⎷', '̲', '➖', '↲', '⩵', '̗', '❢', \n        '≎', '⚔', '⇇', '̑', '⊿', '̖', '☍', '➹', '⥊', '⁁', '✢']\n\ndef clean_punct(x):\n  x = str(x)\n  for punct in puncts:\n    if punct in x:\n      x = x.replace(punct, ' ')\n  return x","metadata":{"execution":{"iopub.status.busy":"2022-01-03T12:15:44.146778Z","iopub.execute_input":"2022-01-03T12:15:44.147066Z","iopub.status.idle":"2022-01-03T12:15:44.322593Z","shell.execute_reply.started":"2022-01-03T12:15:44.147036Z","shell.execute_reply":"2022-01-03T12:15:44.321748Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Correct misspelled words and return some words to their usual form.\n# Ref: https://www.kaggle.com/oysiyl/107-place-solution-using-public-kernel\n\nmispell_dict = {'colour': 'color', 'centre': 'center', 'favourite': 'favorite', 'travelling': 'traveling', 'counselling': 'counseling', 'theatre': 'theater', 'cancelled': 'canceled', 'labour': 'labor', 'organisation': 'organization', 'wwii': 'world war 2', 'citicise': 'criticize', 'youtu ': 'youtube ', 'Qoura': 'Quora', 'sallary': 'salary', 'Whta': 'What', 'narcisist': 'narcissist', 'howdo': 'how do', 'whatare': 'what are', 'howcan': 'how can', 'howmuch': 'how much', 'howmany': 'how many', 'whydo': 'why do', 'doI': 'do I', 'theBest': 'the best', 'howdoes': 'how does', 'mastrubation': 'masturbation', 'mastrubate': 'masturbate', \"mastrubating\": 'masturbating', 'pennis': 'penis', 'Etherium': 'bitcoin', 'narcissit': 'narcissist', 'bigdata': 'big data', '2k17': '2017', '2k18': '2018', 'qouta': 'quota', 'exboyfriend': 'ex boyfriend', 'airhostess': 'air hostess', \"whst\": 'what', 'watsapp': 'whatsapp', 'demonitisation': 'demonetization', 'demonitization': 'demonetization', 'demonetisation': 'demonetization', \n                'electroneum':'bitcoin','nanodegree':'degree','hotstar':'star','dream11':'dream','ftre':'fire','tensorflow':'framework','unocoin':'bitcoin',\n                'lnmiit':'limit','unacademy':'academy','altcoin':'bitcoin','altcoins':'bitcoin','litecoin':'bitcoin','coinbase':'bitcoin','cryptocurency':'cryptocurrency',\n                'simpliv':'simple','quoras':'quora','schizoids':'psychopath','remainers':'remainder','twinflame':'soulmate','quorans':'quora','brexit':'demonetized',\n                'iiest':'institute','dceu':'comics','pessat':'exam','uceed':'college','bhakts':'devotee','boruto':'anime',\n                'cryptocoin':'bitcoin','blockchains':'blockchain','fiancee':'fiance','redmi':'smartphone','oneplus':'smartphone','qoura':'quora','deepmind':'framework','ryzen':'cpu','whattsapp':'whatsapp',\n                'undertale':'adventure','zenfone':'smartphone','cryptocurencies':'cryptocurrencies','koinex':'bitcoin','zebpay':'bitcoin','binance':'bitcoin','whtsapp':'whatsapp',\n                'reactjs':'framework','bittrex':'bitcoin','bitconnect':'bitcoin','bitfinex':'bitcoin','yourquote':'your quote','whyis':'why is','jiophone':'smartphone',\n                'dogecoin':'bitcoin','onecoin':'bitcoin','poloniex':'bitcoin','7700k':'cpu','angular2':'framework','segwit2x':'bitcoin','hashflare':'bitcoin','940mx':'gpu',\n                'openai':'framework','hashflare':'bitcoin','1050ti':'gpu','nearbuy':'near buy','freebitco':'bitcoin','antminer':'bitcoin','filecoin':'bitcoin','whatapp':'whatsapp',\n                'empowr':'empower','1080ti':'gpu','crytocurrency':'cryptocurrency','8700k':'cpu','whatsaap':'whatsapp','g4560':'cpu','payymoney':'pay money',\n                'fuckboys':'fuck boys','intenship':'internship','zcash':'bitcoin','demonatisation':'demonetization','narcicist':'narcissist','mastuburation':'masturbation',\n                'trignometric':'trigonometric','cryptocurreny':'cryptocurrency','howdid':'how did','crytocurrencies':'cryptocurrencies','phycopath':'psychopath',\n                'bytecoin':'bitcoin','possesiveness':'possessiveness','scollege':'college','humanties':'humanities','altacoin':'bitcoin','demonitised':'demonetized',\n                'brasília':'brazilia','accolite':'accolyte','econimics':'economics','varrier':'warrier','quroa':'quora','statergy':'strategy','langague':'language',\n                'splatoon':'game','7600k':'cpu','gate2018':'gate 2018','in2018':'in 2018','narcassist':'narcissist','jiocoin':'bitcoin','hnlu':'hulu','7300hq':'cpu',\n                'weatern':'western','interledger':'blockchain','deplation':'deflation', 'cryptocurrencies':'cryptocurrency', 'bitcoin':'blockchain cryptocurrency',}\n\ndef correct_mispell(x):\n  words = x.split()\n  for i in range(0, len(words)):\n    if mispell_dict.get(words[i]) is not None:\n      words[i] = mispell_dict.get(words[i])\n    elif mispell_dict.get(words[i].lower()) is not None:\n      words[i] = mispell_dict.get(words[i].lower())\n        \n  words = \" \".join(words)\n  return words","metadata":{"execution":{"iopub.status.busy":"2022-01-03T12:15:44.324255Z","iopub.execute_input":"2022-01-03T12:15:44.324579Z","iopub.status.idle":"2022-01-03T12:15:44.34701Z","shell.execute_reply.started":"2022-01-03T12:15:44.324539Z","shell.execute_reply":"2022-01-03T12:15:44.346193Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Remove acronyms\n# Ref: https://www.kaggle.com/theoviel/improve-your-score-with-text-preprocessing-v2 \n\ncontraction_mapping = {\"We'd\": \"We had\", \"That'd\": \"That had\", \"AREN'T\": \"Are not\", \"HADN'T\": \"Had not\", \"Could've\": \"Could have\", \"LeT's\": \"Let us\", \"How'll\": \"How will\", \"They'll\": \"They will\", \"DOESN'T\": \"Does not\", \"HE'S\": \"He has\", \"O'Clock\": \"Of the clock\", \"Who'll\": \"Who will\", \"What'S\": \"What is\", \"Ain't\": \"Am not\", \"WEREN'T\": \"Were not\", \"Y'all\": \"You all\", \"Y'ALL\": \"You all\", \"Here's\": \"Here is\", \"It'd\": \"It had\", \"Should've\": \"Should have\", \"I'M\": \"I am\", \"ISN'T\": \"Is not\", \"Would've\": \"Would have\", \"He'll\": \"He will\", \"DON'T\": \"Do not\", \"She'd\": \"She had\", \"WOULDN'T\": \"Would not\", \"She'll\": \"She will\", \"IT's\": \"It is\", \"There'd\": \"There had\", \"It'll\": \"It will\", \"You'll\": \"You will\", \"He'd\": \"He had\", \"What'll\": \"What will\", \"Ma'am\": \"Madam\", \"CAN'T\": \"Can not\", \"THAT'S\": \"That is\", \"You've\": \"You have\", \"She's\": \"She is\", \"Weren't\": \"Were not\", \"They've\": \"They have\", \"Couldn't\": \"Could not\", \"When's\": \"When is\", \"Haven't\": \"Have not\", \"We'll\": \"We will\", \"That's\": \"That is\", \"We're\": \"We are\", \"They're\": \"They' are\", \"You'd\": \"You would\", \"How'd\": \"How did\", \"What're\": \"What are\", \"Hasn't\": \"Has not\", \"Wasn't\": \"Was not\", \"Won't\": \"Will not\", \"There's\": \"There is\", \"Didn't\": \"Did not\", \"Doesn't\": \"Does not\", \"You're\": \"You are\", \"He's\": \"He is\", \"SO's\": \"So is\", \"We've\": \"We have\", \"Who's\": \"Who is\", \"Wouldn't\": \"Would not\", \"Why's\": \"Why is\", \"WHO's\": \"Who is\", \"Let's\": \"Let us\", \"How's\": \"How is\", \"Can't\": \"Can not\", \"Where's\": \"Where is\", \"They'd\": \"They had\", \"Don't\": \"Do not\", \"Shouldn't\":\"Should not\", \"Aren't\":\"Are not\", \"ain't\": \"is not\", \"What's\": \"What is\", \"It's\": \"It is\", \"Isn't\":\"Is not\", \"aren't\": \"are not\",\"can't\": \"cannot\", \"'cause\": \"because\", \"could've\": \"could have\", \"couldn't\": \"could not\", \"didn't\": \"did not\",  \"doesn't\": \"does not\", \"don't\": \"do not\", \"hadn't\": \"had not\", \"hasn't\": \"has not\", \"haven't\": \"have not\", \"he'd\": \"he would\",\"he'll\": \"he will\", \"he's\": \"he is\", \"how'd\": \"how did\", \"how'd'y\": \"how do you\", \"how'll\": \"how will\", \"how's\": \"how is\",  \"I'd\": \"I would\", \"I'd've\": \"I would have\", \"I'll\": \"I will\", \"I'll've\": \"I will have\",\"I'm\": \"I am\", \"I've\": \"I have\", \"i'd\": \"i would\", \"i'd've\": \"i would have\", \"i'll\": \"i will\",  \"i'll've\": \"i will have\",\"i'm\": \"i am\", \"i've\": \"i have\", \"isn't\": \"is not\", \"it'd\": \"it would\", \"it'd've\": \"it would have\", \"it'll\": \"it will\", \"it'll've\": \"it will have\",\"it's\": \"it is\", \"let's\": \"let us\", \"ma'am\": \"madam\", \"mayn't\": \"may not\", \"might've\": \"might have\",\"mightn't\": \"might not\",\"mightn't've\": \"might not have\", \"must've\": \"must have\", \"mustn't\": \"must not\", \"mustn't've\": \"must not have\", \"needn't\": \"need not\", \"needn't've\": \"need not have\",\"o'clock\": \"of the clock\", \"oughtn't\": \"ought not\", \"oughtn't've\": \"ought not have\", \"shan't\": \"shall not\", \"sha'n't\": \"shall not\", \"shan't've\": \"shall not have\", \"she'd\": \"she would\", \"she'd've\": \"she would have\", \"she'll\": \"she will\", \"she'll've\": \"she will have\", \"she's\": \"she is\", \"should've\": \"should have\", \"shouldn't\": \"should not\", \"shouldn't've\": \"should not have\", \"so've\": \"so have\",\"so's\": \"so as\", \"this's\": \"this is\",\"that'd\": \"that would\", \"that'd've\": \"that would have\", \"that's\": \"that is\", \"there'd\": \"there would\", \"there'd've\": \"there would have\", \"there's\": \"there is\", \"here's\": \"here is\",\"they'd\": \"they would\", \"they'd've\": \"they would have\", \"they'll\": \"they will\", \"they'll've\": \"they will have\", \"they're\": \"they are\", \"they've\": \"they have\", \"to've\": \"to have\", \"wasn't\": \"was not\", \"we'd\": \"we would\", \"we'd've\": \"we would have\", \"we'll\": \"we will\", \"we'll've\": \"we will have\", \"we're\": \"we are\", \"we've\": \"we have\", \"weren't\": \"were not\", \"what'll\": \"what will\", \"what'll've\": \"what will have\", \"what're\": \"what are\",  \"what's\": \"what is\", \"what've\": \"what have\", \"when's\": \"when is\", \"when've\": \"when have\", \"where'd\": \"where did\", \"where's\": \"where is\", \"where've\": \"where have\", \"who'll\": \"who will\", \"who'll've\": \"who will have\", \"who's\": \"who is\", \"who've\": \"who have\", \"why's\": \"why is\", \"why've\": \"why have\", \"will've\": \"will have\", \"won't\": \"will not\", \"won't've\": \"will not have\", \"would've\": \"would have\", \"wouldn't\": \"would not\", \"wouldn't've\": \"would not have\", \"y'all\": \"you all\", \"y'all'd\": \"you all would\",\"y'all'd've\": \"you all would have\",\"y'all're\": \"you all are\",\"y'all've\": \"you all have\",\"you'd\": \"you would\", \"you'd've\": \"you would have\", \"you'll\": \"you will\", \"you'll've\": \"you will have\", \"you're\": \"you are\", \"you've\": \"you have\" }\n\ndef clean_contractions(text):\n    specials = [\"’\", \"‘\", \"´\", \"`\"]\n    for s in specials:\n        text = text.replace(s, \"'\")\n    \n    text = ' '.join([contraction_mapping[t] if t in contraction_mapping else t for t in text.split(\" \")])\n    return text","metadata":{"execution":{"iopub.status.busy":"2022-01-03T12:15:44.350622Z","iopub.execute_input":"2022-01-03T12:15:44.351429Z","iopub.status.idle":"2022-01-03T12:15:44.375039Z","shell.execute_reply.started":"2022-01-03T12:15:44.351394Z","shell.execute_reply":"2022-01-03T12:15:44.373981Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# remove stopword\nstopwords = STOPWORDS  - {'ought', 'whom',\"wouldn't\", \"you'll\", \"you've\"}\n\n\ndef remove_stopwords(x):\n  x = [word for word in x.split() if word not in stopwords]\n  x = ' '.join(x)\n  return x","metadata":{"execution":{"iopub.status.busy":"2022-01-03T12:15:44.376302Z","iopub.execute_input":"2022-01-03T12:15:44.376547Z","iopub.status.idle":"2022-01-03T12:15:44.39036Z","shell.execute_reply.started":"2022-01-03T12:15:44.376518Z","shell.execute_reply":"2022-01-03T12:15:44.389688Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Call all clean-function to clean dataset\ndef data_cleaning(x):\n  x = clean_tag(x)\n  x = clean_punct(x)\n  x = correct_mispell(x)\n  x = clean_contractions(x)\n  x = remove_stopwords(x)\n#   x = lemma_text(x)\n  return x","metadata":{"execution":{"iopub.status.busy":"2022-01-03T12:15:44.392627Z","iopub.execute_input":"2022-01-03T12:15:44.393193Z","iopub.status.idle":"2022-01-03T12:15:44.402657Z","shell.execute_reply.started":"2022-01-03T12:15:44.39315Z","shell.execute_reply":"2022-01-03T12:15:44.401832Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Create cleaned dataset\ntrain['question_text_cleaned'] = train['question_text'].apply(lambda x: data_cleaning(x))\ntest['question_text_cleaned'] = test['question_text'].apply(lambda x: data_cleaning(x))","metadata":{"execution":{"iopub.status.busy":"2022-01-03T12:15:44.403702Z","iopub.execute_input":"2022-01-03T12:15:44.404542Z","iopub.status.idle":"2022-01-03T12:17:11.979627Z","shell.execute_reply.started":"2022-01-03T12:15:44.404501Z","shell.execute_reply":"2022-01-03T12:17:11.978869Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Predict using linearSVC\ndef predict_linearSVC(X_train,y_train,X_test):\n    tfidf.fit(X_train)\n    X_train = tfidf.transform(X_train)\n    X_test = tfidf.transform(X_test)\n    svm = LinearSVC()\n    svm.fit(X_train,y_train)\n    return svm.predict(X_test)\n","metadata":{"execution":{"iopub.status.busy":"2022-01-03T12:17:11.980831Z","iopub.execute_input":"2022-01-03T12:17:11.981637Z","iopub.status.idle":"2022-01-03T12:17:11.987609Z","shell.execute_reply.started":"2022-01-03T12:17:11.981601Z","shell.execute_reply":"2022-01-03T12:17:11.986573Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#create submission file\ndef create_file_submission(predict):\n    submission = pd.DataFrame(test['qid'])\n    submission['prediction'] = predict\n    submission.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-01-03T12:17:11.989629Z","iopub.execute_input":"2022-01-03T12:17:11.990187Z","iopub.status.idle":"2022-01-03T12:17:12.004038Z","shell.execute_reply.started":"2022-01-03T12:17:11.990145Z","shell.execute_reply":"2022-01-03T12:17:12.003324Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def validate_baseModel_dataCleaded():\n    X = train.question_text_cleaned\n    y = train.target\n    X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)\n    predict = predict_linearSVC(X_train,y_train,X_test)\n    return f1_score(predict,y_test)\n\nprint(\"F1-Score in cleaned data: \",validate_baseModel_dataCleaded())","metadata":{"execution":{"iopub.status.busy":"2022-01-03T12:17:12.005885Z","iopub.execute_input":"2022-01-03T12:17:12.006534Z","iopub.status.idle":"2022-01-03T12:20:52.263596Z","shell.execute_reply.started":"2022-01-03T12:17:12.006492Z","shell.execute_reply":"2022-01-03T12:20:52.26217Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def submit():\n    X_train = train['question_text_cleaned']\n    y_train = train.target\n    X_test = test['question_text_cleaned']\n    predict = predict_linearSVC(X_train,y_train,X_test)\n    create_file_submission(predict)\n    \nsubmit()","metadata":{"execution":{"iopub.status.busy":"2022-01-03T12:20:52.266623Z","iopub.execute_input":"2022-01-03T12:20:52.267025Z","iopub.status.idle":"2022-01-03T12:25:41.56047Z","shell.execute_reply.started":"2022-01-03T12:20:52.266963Z","shell.execute_reply":"2022-01-03T12:25:41.559487Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def validate_undersampling():\n    X_target0 = train[train.target == 0].sample(frac = 0.26)\n    data = X_target0.append(train[train.target == 1])\n    X = data.question_text_cleaned\n    y = data.target\n    X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)\n    predict = predict_linearSVC(X_train,y_train,X_test)\n    return f1_score(predict,y_test)\n    \nprint(\"F1-Score with Under Sampling: \",validate_undersampling())","metadata":{"execution":{"iopub.status.busy":"2022-01-03T12:35:44.9007Z","iopub.execute_input":"2022-01-03T12:35:44.901106Z","iopub.status.idle":"2022-01-03T12:36:56.490247Z","shell.execute_reply.started":"2022-01-03T12:35:44.901062Z","shell.execute_reply":"2022-01-03T12:36:56.489464Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def submit_undersampling():\n    X_target0 = train[train.target == 0].sample(frac = 0.26)\n    data = X_target0.append(train[train.target == 1])\n    X_train = data.question_text_cleaned\n    y_train = data.target\n    X_test = test.question_text_cleaned\n    predict = predict_linearSVC(X_train,y_train,X_test)\n    create_file_submission(predict)\n\n#submit_undersampling()","metadata":{"execution":{"iopub.status.busy":"2022-01-03T12:36:56.491638Z","iopub.execute_input":"2022-01-03T12:36:56.492691Z","iopub.status.idle":"2022-01-03T12:36:56.499268Z","shell.execute_reply.started":"2022-01-03T12:36:56.492625Z","shell.execute_reply":"2022-01-03T12:36:56.498713Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def remove_data():\n    \n    train_0 = train[train.target == 0] #Get from class target=0 \n    tfidf.fit(train_0.question_text_cleaned) #encode text with tfidf\n    encode = tfidf.transform(train_0.question_text_cleaned)\n    kmeans = KMeans(n_clusters=14,verbose=1,max_iter=5,n_init=3)\n    kmeans.fit(encode)\n    train_0['cluster'] = kmeans.labels_\n\n    #get 1/4 data\n    a = train_0[train_0.cluster==0] \n    data = a.sample(frac = 0.25)\n    for i in range(1,14):\n        a = train_0[train_0.cluster==i]\n        b = a.sample(frac = 0.25)\n        data = data.append(b)\n\n    data.drop('cluster',axis='columns',inplace=True)\n\n    data = data.append(train[train.target == 1])\n    return data\n\ndata_removed =  remove_data()","metadata":{"execution":{"iopub.status.busy":"2022-01-03T12:25:41.561973Z","iopub.execute_input":"2022-01-03T12:25:41.562208Z","iopub.status.idle":"2022-01-03T12:34:30.723017Z","shell.execute_reply.started":"2022-01-03T12:25:41.56218Z","shell.execute_reply":"2022-01-03T12:34:30.722072Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def validation_undersampling_kmean():\n    X = data_removed.question_text_cleaned\n    y = data_removed.target\n    X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)\n    predict = predict_linearSVC(X_train,y_train,X_test)\n    return f1_score(predict,y_test)\n\nprint(\"F1-Score with Under Sampling and K-mean: \",validation_undersampling_kmean())","metadata":{"execution":{"iopub.status.busy":"2022-01-03T12:34:30.72496Z","iopub.execute_input":"2022-01-03T12:34:30.725294Z","iopub.status.idle":"2022-01-03T12:35:44.891516Z","shell.execute_reply.started":"2022-01-03T12:34:30.725253Z","shell.execute_reply":"2022-01-03T12:35:44.890637Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def submit_undersampling_kmean():\n    X_train = data_removed.question_text_cleaned\n    y_train = data_removed.target\n    X_test = test.question_text_cleaned\n    predict = predict_linearSVC(X_train,y_train,X_test)\n    create_file_submission(predict)\n\n#submit_undersampling_kmean()","metadata":{"execution":{"iopub.status.busy":"2022-01-03T12:35:44.892943Z","iopub.execute_input":"2022-01-03T12:35:44.893259Z","iopub.status.idle":"2022-01-03T12:35:44.898937Z","shell.execute_reply.started":"2022-01-03T12:35:44.893227Z","shell.execute_reply":"2022-01-03T12:35:44.897956Z"},"trusted":true},"execution_count":null,"outputs":[]}]}