{"metadata":{"kernelspec":{"name":"python3","display_name":"Python 3","language":"python"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"colab":{"provenance":[]},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":10737,"databundleVersionId":290346,"sourceType":"competition"}],"dockerImageVersionId":31012,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Libraries","metadata":{}},{"cell_type":"code","source":"import string\nimport pandas as pd\nimport numpy as np\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nimport os\nimport zipfile\n\nfrom sklearn.feature_extraction.text import TfidfVectorizer\nfrom sklearn.svm import LinearSVC\nfrom sklearn.metrics import f1_score\nfrom sklearn.model_selection import train_test_split\nfrom wordcloud import STOPWORDS\nimport seaborn as sns\n# from nltk import WordNetLemmatizer\nimport re\nimport string\nimport os\nimport seaborn as sb\nfrom sklearn.cluster import KMeans\nimport matplotlib.pyplot as plt\nfrom scipy.sparse import coo_matrix, hstack\nfrom sklearn import preprocessing\nfrom tqdm import tqdm","metadata":{"id":"ceDfLG14-J4Z","trusted":true,"execution":{"iopub.status.busy":"2025-05-12T08:53:25.441573Z","iopub.execute_input":"2025-05-12T08:53:25.441938Z","iopub.status.idle":"2025-05-12T08:53:25.449250Z","shell.execute_reply.started":"2025-05-12T08:53:25.441911Z","shell.execute_reply":"2025-05-12T08:53:25.447694Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Load dataset","metadata":{}},{"cell_type":"code","source":"pd.set_option('display.max_colwidth', None)\n\n# Root path\nROOT_PATH = \"/kaggle/input/quora-insincere-questions-classification\"\n\n!ls $ROOT_PATH","metadata":{"id":"O1P9JSgM-TUF","outputId":"46662f1e-0431-4d74-f473-abb2ecb58658","trusted":true,"execution":{"iopub.status.busy":"2025-05-12T08:53:25.450737Z","iopub.execute_input":"2025-05-12T08:53:25.451043Z","iopub.status.idle":"2025-05-12T08:53:25.629495Z","shell.execute_reply.started":"2025-05-12T08:53:25.451018Z","shell.execute_reply":"2025-05-12T08:53:25.628238Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"TRAIN_DF = ROOT_PATH + \"/train.csv\"\n\ntrain_df = pd.read_csv(TRAIN_DF)\ntrain_df","metadata":{"id":"-xJ8wngS-VNT","outputId":"dcc34313-cdbc-4169-f7f3-d92a10cab67b","trusted":true,"execution":{"iopub.status.busy":"2025-05-12T08:53:25.631675Z","iopub.execute_input":"2025-05-12T08:53:25.632008Z","iopub.status.idle":"2025-05-12T08:53:29.020943Z","shell.execute_reply.started":"2025-05-12T08:53:25.631976Z","shell.execute_reply":"2025-05-12T08:53:29.019820Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Data cleaning","metadata":{}},{"cell_type":"code","source":"puncts = [',', '.', '\"', ':', ')', '(', '-', '!', '?', '|', ';', \"'\", '$', '&', '/', '[', ']', '>', '%', '=', '#', '*', '+', '\\\\',\n        '•', '~', '@', '£', '·', '_', '{', '}', '©', '^', '®', '`', '<', '→', '°', '€', '™', '›', '♥', '←', '×', '§', '″', '′',\n        '█', '…', '“', '★', '”', '–', '●', '►', '−', '¢', '¬', '░', '¡', '¶', '↑', '±', '¿', '▾', '═', '¦', '║', '―', '¥', '▓',\n        '—', '‹', '─', '▒', '：', '⊕', '▼', '▪', '†', '■', '’', '▀', '¨', '▄', '♫', '☆', '¯', '♦', '¤', '▲', '¸', '⋅', '‘', '∞',\n        '∙', '）', '↓', '、', '│', '（', '»', '，', '♪', '╩', '╚', '・', '╦', '╣', '╔', '╗', '▬', '❤', '≤', '‡', '√', '◄', '━',\n        '⇒', '▶', '≥', '╝', '♡', '◊', '。', '✈', '≡', '☺', '✔', '↵', '≈', '✓', '♣', '☎', '℃', '◦', '└', '‟', '～', '！', '○',\n        '◆', '№', '♠', '▌', '✿', '▸', '⁄', '□', '❖', '✦', '．', '÷', '｜', '┃', '／', '￥', '╠', '↩', '✭', '▐', '☼', '☻', '┐',\n        '├', '«', '∼', '┌', '℉', '☮', '฿', '≦', '♬', '✧', '〉', '－', '⌂', '✖', '･', '◕', '※', '‖', '◀', '‰', '\\x97', '↺',\n        '∆', '┘', '┬', '╬', '،', '⌘', '⊂', '＞', '〈', '⎙', '？', '☠', '⇐', '▫', '∗', '∈', '≠', '♀', '♔', '˚', '℗', '┗', '＊',\n        '┼', '❀', '＆', '∩', '♂', '‿', '∑', '‣', '➜', '┛', '⇓', '☯', '⊖', '☀', '┳', '；', '∇', '⇑', '✰', '◇', '♯', '☞', '´',\n        '↔', '┏', '｡', '◘', '∂', '✌', '♭', '┣', '┴', '┓', '✨', '\\xa0', '˜', '❥', '┫', '℠', '✒', '［', '∫', '\\x93', '≧', '］',\n        '\\x94', '∀', '♛', '\\x96', '∨', '◎', '↻', '⇩', '＜', '≫', '✩', '✪', '♕', '؟', '₤', '☛', '╮', '␊', '＋', '┈', '％',\n        '╋', '▽', '⇨', '┻', '⊗', '￡', '।', '▂', '✯', '▇', '＿', '➤', '✞', '＝', '▷', '△', '◙', '▅', '✝', '∧', '␉', '☭',\n        '┊', '╯', '☾', '➔', '∴', '\\x92', '▃', '↳', '＾', '׳', '➢', '╭', '➡', '＠', '⊙', '☢', '˝', '∏', '„', '∥', '❝', '☐',\n        '▆', '╱', '⋙', '๏', '☁', '⇔', '▔', '\\x91', '➚', '◡', '╰', '\\x85', '♢', '˙', '۞', '✘', '✮', '☑', '⋆', 'ⓘ', '❒',\n        '☣', '✉', '⌊', '➠', '∣', '❑', '◢', 'ⓒ', '\\x80', '〒', '∕', '▮', '⦿', '✫', '✚', '⋯', '♩', '☂', '❞', '‗', '܂', '☜',\n        '‾', '✜', '╲', '∘', '⟩', '＼', '⟨', '·', '✗', '♚', '∅', 'ⓔ', '◣', '͡', '‛', '❦', '◠', '✄', '❄', '∃', '␣', '≪', '｢',\n        '≅', '◯', '☽', '∎', '｣', '❧', '̅', 'ⓐ', '↘', '⚓', '▣', '˘', '∪', '⇢', '✍', '⊥', '＃', '⎯', '↠', '۩', '☰', '◥',\n        '⊆', '✽', '⚡', '↪', '❁', '☹', '◼', '☃', '◤', '❏', 'ⓢ', '⊱', '➝', '̣', '✡', '∠', '｀', '▴', '┤', '∝', '♏', 'ⓐ',\n        '✎', ';', '␤', '＇', '❣', '✂', '✤', 'ⓞ', '☪', '✴', '⌒', '˛', '♒', '＄', '✶', '▻', 'ⓔ', '◌', '◈', '❚', '❂', '￦',\n        '◉', '╜', '̃', '✱', '╖', '❉', 'ⓡ', '↗', 'ⓣ', '♻', '➽', '׀', '✲', '✬', '☉', '▉', '≒', '☥', '⌐', '♨', '✕', 'ⓝ',\n        '⊰', '❘', '＂', '⇧', '̵', '➪', '▁', '▏', '⊃', 'ⓛ', '‚', '♰', '́', '✏', '⏑', '̶', 'ⓢ', '⩾', '￠', '❍', '≃', '⋰', '♋',\n        '､', '̂', '❋', '✳', 'ⓤ', '╤', '▕', '⌣', '✸', '℮', '⁺', '▨', '╨', 'ⓥ', '♈', '❃', '☝', '✻', '⊇', '≻', '♘', '♞',\n        '◂', '✟', '⌠', '✠', '☚', '✥', '❊', 'ⓒ', '⌈', '❅', 'ⓡ', '♧', 'ⓞ', '▭', '❱', 'ⓣ', '∟', '☕', '♺', '∵', '⍝', 'ⓑ',\n        '✵', '✣', '٭', '♆', 'ⓘ', '∶', '⚜', '◞', '்', '✹', '➥', '↕', '̳', '∷', '✋', '➧', '∋', '̿', 'ͧ', '┅', '⥤', '⬆', '⋱',\n        '☄', '↖', '⋮', '۔', '♌', 'ⓛ', '╕', '♓', '❯', '♍', '▋', '✺', '⭐', '✾', '♊', '➣', '▿', 'ⓑ', '♉', '⏠', '◾', '▹',\n        '⩽', '↦', '╥', '⍵', '⌋', '։', '➨', '∮', '⇥', 'ⓗ', 'ⓓ', '⁻', '⎝', '⌥', '⌉', '◔', '◑', '✼', '♎', '♐', '╪', '⊚',\n        '☒', '⇤', 'ⓜ', '⎠', '◐', '⚠', '╞', '◗', '⎕', 'ⓨ', '☟', 'ⓟ', '♟', '❈', '↬', 'ⓓ', '◻', '♮', '❙', '♤', '∉', '؛',\n        '⁂', 'ⓝ', '־', '♑', '╫', '╓', '╳', '⬅', '☔', '☸', '┄', '╧', '׃', '⎢', '❆', '⋄', '⚫', '̏', '☏', '➞', '͂', '␙',\n        'ⓤ', '◟', '̊', '⚐', '✙', '↙', '̾', '℘', '✷', '⍺', '❌', '⊢', '▵', '✅', 'ⓖ', '☨', '▰', '╡', 'ⓜ', '☤', '∽', '╘',\n        '˹', '↨', '♙', '⬇', '♱', '⌡', '⠀', '╛', '❕', '┉', 'ⓟ', '̀', '♖', 'ⓚ', '┆', '⎜', '◜', '⚾', '⤴', '✇', '╟', '⎛',\n        '☩', '➲', '➟', 'ⓥ', 'ⓗ', '⏝', '◃', '╢', '↯', '✆', '˃', '⍴', '❇', '⚽', '╒', '̸', '♜', '☓', '➳', '⇄', '☬', '⚑',\n        '✐', '⌃', '◅', '▢', '❐', '∊', '☈', '॥', '⎮', '▩', 'ு', '⊹', '‵', '␔', '☊', '➸', '̌', '☿', '⇉', '⊳', '╙', 'ⓦ',\n        '⇣', '｛', '̄', '↝', '⎟', '▍', '❗', '״', '΄', '▞', '◁', '⛄', '⇝', '⎪', '♁', '⇠', '☇', '✊', 'ி', '｝', '⭕', '➘',\n        '⁀', '☙', '❛', '❓', '⟲', '⇀', '≲', 'ⓕ', '⎥', '\\u06dd', 'ͤ', '₋', '̱', '̎', '♝', '≳', '▙', '➭', '܀', 'ⓖ', '⇛', '▊',\n        '⇗', '̷', '⇱', '℅', 'ⓧ', '⚛', '̐', '̕', '⇌', '␀', '≌', 'ⓦ', '⊤', '̓', '☦', 'ⓕ', '▜', '➙', 'ⓨ', '⌨', '◮', '☷',\n        '◍', 'ⓚ', '≔', '⏩', '⍳', '℞', '┋', '˻', '▚', '≺', 'ْ', '▟', '➻', '̪', '⏪', '̉', '⎞', '┇', '⍟', '⇪', '▎', '⇦', '␝',\n        '⤷', '≖', '⟶', '♗', '̴', '♄', 'ͨ', '̈', '❜', '̡', '▛', '✁', '➩', 'ா', '˂', '↥', '⏎', '⎷', '̲', '➖', '↲', '⩵', '̗', '❢',\n        '≎', '⚔', '⇇', '̑', '⊿', '̖', '☍', '➹', '⥊', '⁁', '✢']\n\ndef clean_punct(x):\n  x = str(x)\n  for punct in puncts:\n    if punct in x:\n      x = x.replace(punct, ' ')\n  return x\n\ndef find_unique_words():\n    unique_words = {'a'}\n    non_unique_words = {'a'}\n    for i in tqdm(range(0,len(train_df))):\n        for word in train_df.question_text.iloc[i].split():\n            if word in unique_words:\n                non_unique_words.add(word)\n            else:\n                unique_words.add(word)\n    a = pd.DataFrame(unique_words - non_unique_words)\n    return a\n\nfind_unique_words().head(30)","metadata":{"id":"ll2pOBh5-n4v","outputId":"d7e9af87-e1d3-4e59-be84-98f10a909c43","trusted":true,"execution":{"iopub.status.busy":"2025-05-12T08:53:29.022688Z","iopub.execute_input":"2025-05-12T08:53:29.023351Z","iopub.status.idle":"2025-05-12T08:53:50.236753Z","shell.execute_reply.started":"2025-05-12T08:53:29.023323Z","shell.execute_reply":"2025-05-12T08:53:50.235473Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"mispell_dict = {'colour': 'color', 'centre': 'center', 'favourite': 'favorite', 'travelling': 'traveling', 'counselling': 'counseling', 'theatre': 'theater', 'cancelled': 'canceled', 'labour': 'labor', 'organisation': 'organization', 'wwii': 'world war 2', 'citicise': 'criticize', 'youtu ': 'youtube ', 'Qoura': 'Quora', 'sallary': 'salary', 'Whta': 'What', 'narcisist': 'narcissist', 'howdo': 'how do', 'whatare': 'what are', 'howcan': 'how can', 'howmuch': 'how much', 'howmany': 'how many', 'whydo': 'why do', 'doI': 'do I', 'theBest': 'the best', 'howdoes': 'how does', 'mastrubation': 'masturbation', 'mastrubate': 'masturbate', \"mastrubating\": 'masturbating', 'pennis': 'penis', 'Etherium': 'bitcoin', 'narcissit': 'narcissist', 'bigdata': 'big data', '2k17': '2017', '2k18': '2018', 'qouta': 'quota', 'exboyfriend': 'ex boyfriend', 'airhostess': 'air hostess', \"whst\": 'what', 'watsapp': 'whatsapp', 'demonitisation': 'demonetization', 'demonitization': 'demonetization', 'demonetisation': 'demonetization',\n                'electroneum':'bitcoin','nanodegree':'degree','hotstar':'star','dream11':'dream','ftre':'fire','tensorflow':'framework','unocoin':'bitcoin',\n                'lnmiit':'limit','unacademy':'academy','altcoin':'bitcoin','altcoins':'bitcoin','litecoin':'bitcoin','coinbase':'bitcoin','cryptocurency':'cryptocurrency',\n                'simpliv':'simple','quoras':'quora','schizoids':'psychopath','remainers':'remainder','twinflame':'soulmate','quorans':'quora','brexit':'demonetized',\n                'iiest':'institute','dceu':'comics','pessat':'exam','uceed':'college','bhakts':'devotee','boruto':'anime',\n                'cryptocoin':'bitcoin','blockchains':'blockchain','fiancee':'fiance','redmi':'smartphone','oneplus':'smartphone','qoura':'quora','deepmind':'framework','ryzen':'cpu','whattsapp':'whatsapp',\n                'undertale':'adventure','zenfone':'smartphone','cryptocurencies':'cryptocurrencies','koinex':'bitcoin','zebpay':'bitcoin','binance':'bitcoin','whtsapp':'whatsapp',\n                'reactjs':'framework','bittrex':'bitcoin','bitconnect':'bitcoin','bitfinex':'bitcoin','yourquote':'your quote','whyis':'why is','jiophone':'smartphone',\n                'dogecoin':'bitcoin','onecoin':'bitcoin','poloniex':'bitcoin','7700k':'cpu','angular2':'framework','segwit2x':'bitcoin','hashflare':'bitcoin','940mx':'gpu',\n                'openai':'framework','hashflare':'bitcoin','1050ti':'gpu','nearbuy':'near buy','freebitco':'bitcoin','antminer':'bitcoin','filecoin':'bitcoin','whatapp':'whatsapp',\n                'empowr':'empower','1080ti':'gpu','crytocurrency':'cryptocurrency','8700k':'cpu','whatsaap':'whatsapp','g4560':'cpu','payymoney':'pay money',\n                'fuckboys':'fuck boys','intenship':'internship','zcash':'bitcoin','demonatisation':'demonetization','narcicist':'narcissist','mastuburation':'masturbation',\n                'trignometric':'trigonometric','cryptocurreny':'cryptocurrency','howdid':'how did','crytocurrencies':'cryptocurrencies','phycopath':'psychopath',\n                'bytecoin':'bitcoin','possesiveness':'possessiveness','scollege':'college','humanties':'humanities','altacoin':'bitcoin','demonitised':'demonetized',\n                'brasília':'brazilia','accolite':'accolyte','econimics':'economics','varrier':'warrier','quroa':'quora','statergy':'strategy','langague':'language',\n                'splatoon':'game','7600k':'cpu','gate2018':'gate 2018','in2018':'in 2018','narcassist':'narcissist','jiocoin':'bitcoin','hnlu':'hulu','7300hq':'cpu',\n                'weatern':'western','interledger':'blockchain','deplation':'deflation', 'cryptocurrencies':'cryptocurrency', 'bitcoin':'blockchain cryptocurrency',}\n\ndef correct_mispell(x):\n  words = x.split()\n  for i in range(0, len(words)):\n    if mispell_dict.get(words[i]) is not None:\n      words[i] = mispell_dict.get(words[i])\n    elif mispell_dict.get(words[i].lower()) is not None:\n      words[i] = mispell_dict.get(words[i].lower())\n\n  words = \" \".join(words)\n  return words","metadata":{"id":"EKFJQhsF-uy7","trusted":true,"execution":{"iopub.status.busy":"2025-05-12T08:53:50.238457Z","iopub.execute_input":"2025-05-12T08:53:50.238758Z","iopub.status.idle":"2025-05-12T08:53:50.257202Z","shell.execute_reply.started":"2025-05-12T08:53:50.238733Z","shell.execute_reply":"2025-05-12T08:53:50.255740Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"contraction_mapping = {\"We'd\": \"We had\", \"That'd\": \"That had\", \"AREN'T\": \"Are not\", \"HADN'T\": \"Had not\", \"Could've\": \"Could have\", \"LeT's\": \"Let us\", \"How'll\": \"How will\", \"They'll\": \"They will\", \"DOESN'T\": \"Does not\", \"HE'S\": \"He has\", \"O'Clock\": \"Of the clock\", \"Who'll\": \"Who will\", \"What'S\": \"What is\", \"Ain't\": \"Am not\", \"WEREN'T\": \"Were not\", \"Y'all\": \"You all\", \"Y'ALL\": \"You all\", \"Here's\": \"Here is\", \"It'd\": \"It had\", \"Should've\": \"Should have\", \"I'M\": \"I am\", \"ISN'T\": \"Is not\", \"Would've\": \"Would have\", \"He'll\": \"He will\", \"DON'T\": \"Do not\", \"She'd\": \"She had\", \"WOULDN'T\": \"Would not\", \"She'll\": \"She will\", \"IT's\": \"It is\", \"There'd\": \"There had\", \"It'll\": \"It will\", \"You'll\": \"You will\", \"He'd\": \"He had\", \"What'll\": \"What will\", \"Ma'am\": \"Madam\", \"CAN'T\": \"Can not\", \"THAT'S\": \"That is\", \"You've\": \"You have\", \"She's\": \"She is\", \"Weren't\": \"Were not\", \"They've\": \"They have\", \"Couldn't\": \"Could not\", \"When's\": \"When is\", \"Haven't\": \"Have not\", \"We'll\": \"We will\", \"That's\": \"That is\", \"We're\": \"We are\", \"They're\": \"They' are\", \"You'd\": \"You would\", \"How'd\": \"How did\", \"What're\": \"What are\", \"Hasn't\": \"Has not\", \"Wasn't\": \"Was not\", \"Won't\": \"Will not\", \"There's\": \"There is\", \"Didn't\": \"Did not\", \"Doesn't\": \"Does not\", \"You're\": \"You are\", \"He's\": \"He is\", \"SO's\": \"So is\", \"We've\": \"We have\", \"Who's\": \"Who is\", \"Wouldn't\": \"Would not\", \"Why's\": \"Why is\", \"WHO's\": \"Who is\", \"Let's\": \"Let us\", \"How's\": \"How is\", \"Can't\": \"Can not\", \"Where's\": \"Where is\", \"They'd\": \"They had\", \"Don't\": \"Do not\", \"Shouldn't\":\"Should not\", \"Aren't\":\"Are not\", \"ain't\": \"is not\", \"What's\": \"What is\", \"It's\": \"It is\", \"Isn't\":\"Is not\", \"aren't\": \"are not\",\"can't\": \"cannot\", \"'cause\": \"because\", \"could've\": \"could have\", \"couldn't\": \"could not\", \"didn't\": \"did not\",  \"doesn't\": \"does not\", \"don't\": \"do not\", \"hadn't\": \"had not\", \"hasn't\": \"has not\", \"haven't\": \"have not\", \"he'd\": \"he would\",\"he'll\": \"he will\", \"he's\": \"he is\", \"how'd\": \"how did\", \"how'd'y\": \"how do you\", \"how'll\": \"how will\", \"how's\": \"how is\",  \"I'd\": \"I would\", \"I'd've\": \"I would have\", \"I'll\": \"I will\", \"I'll've\": \"I will have\",\"I'm\": \"I am\", \"I've\": \"I have\", \"i'd\": \"i would\", \"i'd've\": \"i would have\", \"i'll\": \"i will\",  \"i'll've\": \"i will have\",\"i'm\": \"i am\", \"i've\": \"i have\", \"isn't\": \"is not\", \"it'd\": \"it would\", \"it'd've\": \"it would have\", \"it'll\": \"it will\", \"it'll've\": \"it will have\",\"it's\": \"it is\", \"let's\": \"let us\", \"ma'am\": \"madam\", \"mayn't\": \"may not\", \"might've\": \"might have\",\"mightn't\": \"might not\",\"mightn't've\": \"might not have\", \"must've\": \"must have\", \"mustn't\": \"must not\", \"mustn't've\": \"must not have\", \"needn't\": \"need not\", \"needn't've\": \"need not have\",\"o'clock\": \"of the clock\", \"oughtn't\": \"ought not\", \"oughtn't've\": \"ought not have\", \"shan't\": \"shall not\", \"sha'n't\": \"shall not\", \"shan't've\": \"shall not have\", \"she'd\": \"she would\", \"she'd've\": \"she would have\", \"she'll\": \"she will\", \"she'll've\": \"she will have\", \"she's\": \"she is\", \"should've\": \"should have\", \"shouldn't\": \"should not\", \"shouldn't've\": \"should not have\", \"so've\": \"so have\",\"so's\": \"so as\", \"this's\": \"this is\",\"that'd\": \"that would\", \"that'd've\": \"that would have\", \"that's\": \"that is\", \"there'd\": \"there would\", \"there'd've\": \"there would have\", \"there's\": \"there is\", \"here's\": \"here is\",\"they'd\": \"they would\", \"they'd've\": \"they would have\", \"they'll\": \"they will\", \"they'll've\": \"they will have\", \"they're\": \"they are\", \"they've\": \"they have\", \"to've\": \"to have\", \"wasn't\": \"was not\", \"we'd\": \"we would\", \"we'd've\": \"we would have\", \"we'll\": \"we will\", \"we'll've\": \"we will have\", \"we're\": \"we are\", \"we've\": \"we have\", \"weren't\": \"were not\", \"what'll\": \"what will\", \"what'll've\": \"what will have\", \"what're\": \"what are\",  \"what's\": \"what is\", \"what've\": \"what have\", \"when's\": \"when is\", \"when've\": \"when have\", \"where'd\": \"where did\", \"where's\": \"where is\", \"where've\": \"where have\", \"who'll\": \"who will\", \"who'll've\": \"who will have\", \"who's\": \"who is\", \"who've\": \"who have\", \"why's\": \"why is\", \"why've\": \"why have\", \"will've\": \"will have\", \"won't\": \"will not\", \"won't've\": \"will not have\", \"would've\": \"would have\", \"wouldn't\": \"would not\", \"wouldn't've\": \"would not have\", \"y'all\": \"you all\", \"y'all'd\": \"you all would\",\"y'all'd've\": \"you all would have\",\"y'all're\": \"you all are\",\"y'all've\": \"you all have\",\"you'd\": \"you would\", \"you'd've\": \"you would have\", \"you'll\": \"you will\", \"you'll've\": \"you will have\", \"you're\": \"you are\", \"you've\": \"you have\" }\n\ndef clean_contractions(text):\n    specials = [\"’\", \"‘\", \"´\", \"`\"]\n    for s in specials:\n        text = text.replace(s, \"'\")\n\n    text = ' '.join([contraction_mapping[t] if t in contraction_mapping else t for t in text.split(\" \")])\n    return text","metadata":{"id":"iIWA2sJM-xho","trusted":true,"execution":{"iopub.status.busy":"2025-05-12T08:53:50.259403Z","iopub.execute_input":"2025-05-12T08:53:50.260000Z","iopub.status.idle":"2025-05-12T08:53:50.289993Z","shell.execute_reply.started":"2025-05-12T08:53:50.259968Z","shell.execute_reply":"2025-05-12T08:53:50.288871Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"stopwords = STOPWORDS  - {'ought', 'whom',\"wouldn't\", \"you'll\", \"you've\"}# bỏ đi một số từ khiến hiệu quả dự đoán giảm\n\n\ndef remove_stopwords(x):\n  x = [word for word in x.split() if word not in stopwords]\n  x = ' '.join(x)\n  return x","metadata":{"id":"8_AVEomW-zZy","trusted":true,"execution":{"iopub.status.busy":"2025-05-12T08:53:50.291278Z","iopub.execute_input":"2025-05-12T08:53:50.291726Z","iopub.status.idle":"2025-05-12T08:53:50.318711Z","shell.execute_reply.started":"2025-05-12T08:53:50.291692Z","shell.execute_reply":"2025-05-12T08:53:50.317829Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Lấy từ stopword đã bỏ trong câu\ndef UncommonWords(A, B):\n\n    count = {'a'}\n\n    # insert words of string A to hash\n    for word in A.split():\n        if word not in B:\n            count.add(word)\n    # return required list of words\n\n    return count","metadata":{"id":"eQT79lLR-1ap","trusted":true,"execution":{"iopub.status.busy":"2025-05-12T08:53:50.319789Z","iopub.execute_input":"2025-05-12T08:53:50.320175Z","iopub.status.idle":"2025-05-12T08:53:50.340704Z","shell.execute_reply.started":"2025-05-12T08:53:50.320121Z","shell.execute_reply":"2025-05-12T08:53:50.339566Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def data_cleaning(x):\n  x = clean_punct(x)\n  x = correct_mispell(x)\n  x = clean_contractions(x)\n  x = remove_stopwords(x)\n#   x = lemma_text(x)\n  return x","metadata":{"id":"gATa6UOa-4Z1","trusted":true,"execution":{"iopub.status.busy":"2025-05-12T08:53:50.341913Z","iopub.execute_input":"2025-05-12T08:53:50.342246Z","iopub.status.idle":"2025-05-12T08:53:50.365903Z","shell.execute_reply.started":"2025-05-12T08:53:50.342220Z","shell.execute_reply":"2025-05-12T08:53:50.364803Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Tạo ra 1 tập dữ liệu đã được cleaning\ntrain_df['question_text_cleaned'] = train_df['question_text'].apply(lambda x: data_cleaning(x))","metadata":{"id":"2ERy85AY-6bC","trusted":true,"execution":{"iopub.status.busy":"2025-05-12T08:53:50.367069Z","iopub.execute_input":"2025-05-12T08:53:50.367500Z","iopub.status.idle":"2025-05-12T08:54:35.717181Z","shell.execute_reply.started":"2025-05-12T08:53:50.367474Z","shell.execute_reply":"2025-05-12T08:54:35.716198Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install unidecode","metadata":{"id":"3CeilDy3_BfP","outputId":"3c8cd7ec-dd7d-4dea-fea9-81dd09325911","trusted":true,"execution":{"iopub.status.busy":"2025-05-12T08:54:35.718471Z","iopub.execute_input":"2025-05-12T08:54:35.718770Z","iopub.status.idle":"2025-05-12T08:54:39.381475Z","shell.execute_reply.started":"2025-05-12T08:54:35.718745Z","shell.execute_reply":"2025-05-12T08:54:39.380237Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from unidecode import unidecode\n\ndef clean(text: str) -> str:\n    # Chuyển đổi ký tự Unicode thành dạng ASCII\n    uni_text = str(unidecode(text).encode(\"ascii\"), \"ascii\")\n\n    # Chuyển văn bản thành chữ thường\n    lower = uni_text.lower()\n\n    return lower\n","metadata":{"id":"nZlMeaFx_DXK","trusted":true,"execution":{"iopub.status.busy":"2025-05-12T08:54:39.385717Z","iopub.execute_input":"2025-05-12T08:54:39.386053Z","iopub.status.idle":"2025-05-12T08:54:39.391900Z","shell.execute_reply.started":"2025-05-12T08:54:39.386025Z","shell.execute_reply":"2025-05-12T08:54:39.390860Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Clean data\ninput = []\nfor idx, ques in enumerate(train_df['question_text_cleaned']):\n  input.append(clean(ques))","metadata":{"id":"iG927VVs_GLe","trusted":true,"execution":{"iopub.status.busy":"2025-05-12T08:54:39.393105Z","iopub.execute_input":"2025-05-12T08:54:39.393526Z","iopub.status.idle":"2025-05-12T08:54:40.673294Z","shell.execute_reply.started":"2025-05-12T08:54:39.393494Z","shell.execute_reply":"2025-05-12T08:54:40.672343Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df['question_text_cleaned'] = input\ntrain_df","metadata":{"id":"WI5-zfVo_IJV","outputId":"6f65d3d2-ccc4-470a-e800-10d6922b95e7","trusted":true,"execution":{"iopub.status.busy":"2025-05-12T08:54:40.674318Z","iopub.execute_input":"2025-05-12T08:54:40.674581Z","iopub.status.idle":"2025-05-12T08:54:40.831556Z","shell.execute_reply.started":"2025-05-12T08:54:40.674561Z","shell.execute_reply":"2025-05-12T08:54:40.830540Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df.to_csv('train_df.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-12T08:54:40.832721Z","iopub.execute_input":"2025-05-12T08:54:40.832977Z","iopub.status.idle":"2025-05-12T08:54:49.046247Z","shell.execute_reply.started":"2025-05-12T08:54:40.832957Z","shell.execute_reply":"2025-05-12T08:54:49.045191Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Model","metadata":{}},{"cell_type":"code","source":"df = pd.read_csv(\"/kaggle/working/train_df.csv\")\ndf = df.dropna(subset=['question_text_cleaned'])\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-12T08:54:49.047161Z","iopub.execute_input":"2025-05-12T08:54:49.047539Z","iopub.status.idle":"2025-05-12T08:54:54.844768Z","shell.execute_reply.started":"2025-05-12T08:54:49.047497Z","shell.execute_reply":"2025-05-12T08:54:54.843643Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import scipy.sparse as sp\nfrom sklearn.feature_extraction.text import TfidfVectorizer\nfrom sklearn.utils import resample\n\n# Separate the data into two classes (sincere and insincere)\nsincere_df = df[df['target'] == 0]  # Sincere class\ninsincere_df = df[df['target'] == 1]  # Insincere class\n\n\nsincere_downsampled = resample(sincere_df, \n                               replace=False,    # Do not resample with replacement\n                               n_samples=len(insincere_df),  # Match the number of insincere samples\n                               random_state=42)  # Ensure reproducibility\n\n# Combine the downsampled sincere class with the insincere class\nbalanced_df = pd.concat([sincere_downsampled, insincere_df])\n\n# Now, shuffle the dataset (if needed) and sample the 30,000 balanced data points\nbalanced_df = balanced_df.sample(n=30000, random_state=42)\n\n# Select the cleaned question texts and labels\nquestions = balanced_df['question_text_cleaned'].tolist()\nlabels = balanced_df['target'].tolist()\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-12T08:54:54.845849Z","iopub.execute_input":"2025-05-12T08:54:54.846149Z","iopub.status.idle":"2025-05-12T08:54:55.253510Z","shell.execute_reply.started":"2025-05-12T08:54:54.846106Z","shell.execute_reply":"2025-05-12T08:54:55.252171Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"balanced_df['target'].value_counts()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-12T08:54:55.254471Z","iopub.execute_input":"2025-05-12T08:54:55.254758Z","iopub.status.idle":"2025-05-12T08:54:55.269215Z","shell.execute_reply.started":"2025-05-12T08:54:55.254735Z","shell.execute_reply":"2025-05-12T08:54:55.268277Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport scipy.sparse as sp\nimport torch\nfrom sklearn.feature_extraction.text import TfidfVectorizer\n# === TF-IDF with smaller vocabulary ===\nvectorizer = TfidfVectorizer(max_features=5000)\ntfidf = vectorizer.fit_transform(questions)\nvocab = vectorizer.get_feature_names_out()\n\nnum_docs = len(questions)\nnum_words = len(vocab)\n\n# === Build adjacency matrix in COO format ===\nrow, col, data = [], [], []\n\nfor doc_id in range(num_docs):\n    for word_id in tfidf[doc_id].nonzero()[1]:\n        val = tfidf[doc_id, word_id]\n        word_node = num_docs + word_id\n        row.extend([doc_id, word_node])\n        col.extend([word_node, doc_id])\n        data.extend([val, val])\n\n# Add identity/self-loops\nidentity = np.arange(num_docs + num_words)\nrow.extend(identity)\ncol.extend(identity)\ndata.extend([1] * len(identity))\n\n# COO sparse matrix\nadj = sp.coo_matrix((data, (row, col)), shape=(num_docs + num_words, num_docs + num_words))\n\n# === Convert to PyTorch Sparse Tensor ===\ndef sparse_mx_to_torch_sparse_tensor(sparse_mx):\n    sparse_mx = sparse_mx.tocoo().astype(np.float32)\n    indices = torch.from_numpy(np.vstack((sparse_mx.row, sparse_mx.col)).astype(np.int64))\n    values = torch.from_numpy(sparse_mx.data)\n    shape = torch.Size(sparse_mx.shape)\n    return torch.sparse.FloatTensor(indices, values, shape)\n\nadj = sparse_mx_to_torch_sparse_tensor(adj)\n\n# Use dense identity as features\nfeatures = torch.eye(num_docs + num_words)\nlabels = torch.LongTensor(labels)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-12T08:54:55.270224Z","iopub.execute_input":"2025-05-12T08:54:55.270542Z","iopub.status.idle":"2025-05-12T08:55:12.038530Z","shell.execute_reply.started":"2025-05-12T08:54:55.270510Z","shell.execute_reply":"2025-05-12T08:55:12.037359Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import torch.nn as nn\nimport torch.nn.functional as F\n\nclass SparseTextGCN(nn.Module):\n    def __init__(self, input_dim, hidden_dim, num_classes, dropout=0.5):\n        super(SparseTextGCN, self).__init__()\n        self.fc1 = nn.Linear(input_dim, hidden_dim)\n        self.fc2 = nn.Linear(hidden_dim, num_classes)\n        self.dropout = dropout\n\n    def forward(self, x, adj):\n        x = torch.spmm(adj, x)\n        x = F.relu(self.fc1(x))\n        x = F.dropout(x, self.dropout, training=self.training)\n        x = torch.spmm(adj, x)\n        x = self.fc2(x)\n        return x\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-12T08:55:12.039766Z","iopub.execute_input":"2025-05-12T08:55:12.040356Z","iopub.status.idle":"2025-05-12T08:55:12.048538Z","shell.execute_reply.started":"2025-05-12T08:55:12.040311Z","shell.execute_reply":"2025-05-12T08:55:12.046870Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\ntrain_idx, val_idx, y_train, y_val = train_test_split(\n    np.arange(num_docs), labels, test_size=0.2, random_state=42, stratify=labels\n)\n\ntrain_idx = torch.LongTensor(train_idx)\nval_idx = torch.LongTensor(val_idx)\ny_train = torch.LongTensor(y_train)\ny_val = torch.LongTensor(y_val)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-12T08:55:12.049827Z","iopub.execute_input":"2025-05-12T08:55:12.050222Z","iopub.status.idle":"2025-05-12T08:55:12.117781Z","shell.execute_reply.started":"2025-05-12T08:55:12.050190Z","shell.execute_reply":"2025-05-12T08:55:12.116678Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model = SparseTextGCN(input_dim=features.shape[1], hidden_dim=128, num_classes=2)\noptimizer = torch.optim.Adam(model.parameters(), lr=0.01)\ncriterion = nn.CrossEntropyLoss()\ntrain_mask = torch.arange(num_docs)  # Only documents have labels\n\nfor epoch in range(10):\n    model.train()\n    optimizer.zero_grad()\n    output = model(features, adj)\n    loss = criterion(output[train_idx], y_train)\n    loss.backward()\n    optimizer.step()\n\n    # Optional: Evaluate during training\n    model.eval()\n    with torch.no_grad():\n        val_logits = output[val_idx]\n        val_preds = torch.argmax(val_logits, dim=1)\n        val_acc = (val_preds == y_val).float().mean().item()\n    print(f\"Epoch {epoch}, Loss: {loss.item():.4f}, Val Acc: {val_acc:.4f}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-12T08:55:12.118922Z","iopub.execute_input":"2025-05-12T08:55:12.119260Z","iopub.status.idle":"2025-05-12T08:58:10.820092Z","shell.execute_reply.started":"2025-05-12T08:55:12.119235Z","shell.execute_reply":"2025-05-12T08:58:10.819093Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Save the model\ntorch.save(model.state_dict(), \"textgcn_model.pth\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-12T08:58:10.821522Z","iopub.execute_input":"2025-05-12T08:58:10.822269Z","iopub.status.idle":"2025-05-12T08:58:10.848406Z","shell.execute_reply.started":"2025-05-12T08:58:10.822239Z","shell.execute_reply":"2025-05-12T08:58:10.847283Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.metrics import classification_report\n\n# After model evaluation\nmodel.eval()\nwith torch.no_grad():\n    output = model(features, adj)\n    logits = output[val_idx]\n    preds = torch.argmax(logits, dim=1)\n\n# Generate the classification report\nreport = classification_report(y_val.cpu(), preds.cpu(), target_names=[\"Sincere\", \"Insincere\"])\nprint(report)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-12T08:58:10.850021Z","iopub.execute_input":"2025-05-12T08:58:10.850986Z","iopub.status.idle":"2025-05-12T08:58:25.072842Z","shell.execute_reply.started":"2025-05-12T08:58:10.850954Z","shell.execute_reply":"2025-05-12T08:58:25.071997Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nfrom sklearn.metrics import confusion_matrix, ConfusionMatrixDisplay\n\n# After model evaluation\nmodel.eval()\nwith torch.no_grad():\n    output = model(features, adj)\n    logits = output[val_idx]\n    preds = torch.argmax(logits, dim=1)\n\n# Compute confusion matrix\ncm = confusion_matrix(y_val.cpu(), preds.cpu())\ndisp = ConfusionMatrixDisplay(confusion_matrix=cm, display_labels=[\"Sincere\", \"Insincere\"])\n\n# Plot\nplt.figure(figsize=(6, 5))\ndisp.plot(cmap=\"Blues\", values_format='d')\nplt.title(\"Confusion Matrix (Validation Set)\")\nplt.grid(False)\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-12T08:58:25.073714Z","iopub.execute_input":"2025-05-12T08:58:25.073943Z","iopub.status.idle":"2025-05-12T08:58:39.465005Z","shell.execute_reply.started":"2025-05-12T08:58:25.073925Z","shell.execute_reply":"2025-05-12T08:58:39.463883Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.manifold import TSNE\nimport matplotlib.pyplot as plt\n\n# Extract document embeddings (only first num_docs are documents)\ndoc_embeddings = output[:num_docs].detach().cpu().numpy()\n\n# Reduce to 2D\ntsne = TSNE(n_components=2, random_state=42)\ndoc_embeddings_2d = tsne.fit_transform(doc_embeddings)\n\n# Create a scatter plot\nplt.figure(figsize=(10, 7))\ncolors = ['blue' if label == 0 else 'red' for label in labels]\nplt.scatter(doc_embeddings_2d[:, 0], doc_embeddings_2d[:, 1], c=colors, alpha=0.5, s=10)\nplt.title(\"t-SNE Visualization of Document Embeddings\")\nplt.xlabel(\"X1\")\nplt.ylabel(\"X2\")\nplt.legend(handles=[\n    plt.Line2D([0], [0], marker='o', color='w', label='Sincere', markerfacecolor='blue', markersize=8),\n    plt.Line2D([0], [0], marker='o', color='w', label='Insincere', markerfacecolor='red', markersize=8)\n])\nplt.grid(True)\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-12T08:58:39.466210Z","iopub.execute_input":"2025-05-12T08:58:39.466577Z","iopub.status.idle":"2025-05-12T09:03:26.046740Z","shell.execute_reply.started":"2025-05-12T08:58:39.466537Z","shell.execute_reply":"2025-05-12T09:03:26.045357Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.manifold import TSNE\nimport matplotlib.pyplot as plt\nimport numpy as np\n\n# Extract document embeddings and corresponding labels\ndoc_embeddings = output[:num_docs].detach().cpu().numpy()\ndoc_labels = labels[:num_docs]  # Ensure slicing matches only document nodes\n\n# Randomly sample 1000 documents\nnp.random.seed(42)\nsample_indices = np.random.choice(num_docs, size=1000, replace=False)\nsampled_embeddings = doc_embeddings[sample_indices]\nsampled_labels = np.array(doc_labels)[sample_indices]\n\n# Apply t-SNE to the sampled embeddings\ntsne = TSNE(n_components=2, random_state=42)\nembeddings_2d = tsne.fit_transform(sampled_embeddings)\n\n# Plot\nplt.figure(figsize=(10, 7))\ncolors = ['blue' if label == 0 else 'red' for label in sampled_labels]\nplt.scatter(embeddings_2d[:, 0], embeddings_2d[:, 1], c=colors, alpha=0.6, s=12)\nplt.title(\"t-SNE visualization of GCN embeddings for Quora dataset\")\nplt.xlabel(\"X1\")\nplt.ylabel(\"X2\")\nplt.legend(handles=[\n    plt.Line2D([0], [0], marker='o', color='w', label='Sincere', markerfacecolor='blue', markersize=8),\n    plt.Line2D([0], [0], marker='o', color='w', label='Insincere', markerfacecolor='red', markersize=8)\n])\nplt.grid(True)\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-12T09:03:26.047926Z","iopub.execute_input":"2025-05-12T09:03:26.048316Z","iopub.status.idle":"2025-05-12T09:03:31.387599Z","shell.execute_reply.started":"2025-05-12T09:03:26.048280Z","shell.execute_reply":"2025-05-12T09:03:31.386107Z"}},"outputs":[],"execution_count":null}]}