{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# **Introduction**\nClassifying Quora questions whether they are insincere or sincere ones","metadata":{}},{"cell_type":"markdown","source":"# **Import necessary libraries**\n# **Import spacy**","metadata":{}},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n\n%matplotlib inline\nimport warnings\nwarnings.filterwarnings('ignore')\n\n#import necessary libraries\nimport os\nimport csv\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport string\n\n#import spacy\nimport re\nimport nltk\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom PIL import Image\nfrom wordcloud import WordCloud #tag cloud: novelty visual representation of text data","metadata":{"execution":{"iopub.status.busy":"2021-09-27T14:21:29.049718Z","iopub.execute_input":"2021-09-27T14:21:29.05003Z","iopub.status.idle":"2021-09-27T14:21:30.416384Z","shell.execute_reply.started":"2021-09-27T14:21:29.049956Z","shell.execute_reply":"2021-09-27T14:21:30.415529Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Starting to understand the input files**","metadata":{}},{"cell_type":"markdown","source":"# **Load data into dataframe then print out to observe**","metadata":{}},{"cell_type":"markdown","source":"Read input files <br>\nUsing pandas: CSV file input/output","metadata":{}},{"cell_type":"code","source":"train_data = pd.read_csv('../input/quora-insincere-questions-classification/train.csv')\ntest_data = pd.read_csv('../input/quora-insincere-questions-classification/test.csv')","metadata":{"execution":{"iopub.status.busy":"2021-09-27T14:21:30.419594Z","iopub.execute_input":"2021-09-27T14:21:30.419848Z","iopub.status.idle":"2021-09-27T14:21:35.543163Z","shell.execute_reply.started":"2021-09-27T14:21:30.419822Z","shell.execute_reply":"2021-09-27T14:21:35.542345Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Print out some of the first data in train.csv file**\n+ The raw data contains 1306122 rows and 3 columns <br>\n+ The feature includes \"questions id\", \"questions text\", \"target\" <br> <br>\n+ Questions id: Id of the question, qid may not take part in classfying questions -> can ignore <br>\n+ Questions text: Since this field is the only one that directly affects the subclass of the question, preprocessing is required <br>\n+ Target: Sincere question target = 0; Insincere question target = 1<br> <br>\n+ Can not see the questions classification yet","metadata":{}},{"cell_type":"code","source":"train_data.head()","metadata":{"execution":{"iopub.status.busy":"2021-09-27T14:21:35.545799Z","iopub.execute_input":"2021-09-27T14:21:35.546403Z","iopub.status.idle":"2021-09-27T14:21:35.568158Z","shell.execute_reply.started":"2021-09-27T14:21:35.546363Z","shell.execute_reply":"2021-09-27T14:21:35.567398Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Print out some of the first data in test.csv file**\n+ The raw data contains 375806 rows and 2 columns <br>\n+ The feature includes questions id, questions text <br> <br>\n--> These questions are the ones we have to set target (0, 1), which is the purpose of this challenge","metadata":{}},{"cell_type":"code","source":"test_data.head()","metadata":{"execution":{"iopub.status.busy":"2021-09-27T14:21:35.57045Z","iopub.execute_input":"2021-09-27T14:21:35.571095Z","iopub.status.idle":"2021-09-27T14:21:35.58116Z","shell.execute_reply.started":"2021-09-27T14:21:35.57098Z","shell.execute_reply":"2021-09-27T14:21:35.580066Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Dimensions of Training Dataset : \", train_data.shape)\nprint(\"Dimensions of Test Dataset : \", test_data.shape)","metadata":{"execution":{"iopub.status.busy":"2021-09-27T14:21:35.582384Z","iopub.execute_input":"2021-09-27T14:21:35.582718Z","iopub.status.idle":"2021-09-27T14:21:35.590034Z","shell.execute_reply.started":"2021-09-27T14:21:35.582676Z","shell.execute_reply":"2021-09-27T14:21:35.589086Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.countplot(x=\"target\", data=train_data, palette=\"Set1\")\nplt.title('Target Count')","metadata":{"execution":{"iopub.status.busy":"2021-09-27T14:21:35.591409Z","iopub.execute_input":"2021-09-27T14:21:35.591776Z","iopub.status.idle":"2021-09-27T14:21:35.786759Z","shell.execute_reply.started":"2021-09-27T14:21:35.59174Z","shell.execute_reply":"2021-09-27T14:21:35.785982Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"--> The number of sincere questions are much greater than the number of insincere questions","metadata":{}},{"cell_type":"markdown","source":"# **Print out the number of sincere questions and insincere questions**\n+ Sincere questions have the target tag = 0 <br>\n+ Insincere questions have the target tag = 1","metadata":{}},{"cell_type":"code","source":"num_questions = len(train_data['qid'])\nnum_sincere_questions = len(train_data.qid[train_data['target'] == 0])\nnum_insincere_questions = len(train_data.qid[train_data['target'] == 1])\nprint(\"Number of Sincere questions in the training set : \", num_sincere_questions)\nprint(\"Number of Insincere questions in the training set : \", num_insincere_questions)","metadata":{"execution":{"iopub.status.busy":"2021-09-27T14:22:12.959159Z","iopub.execute_input":"2021-09-27T14:22:12.959571Z","iopub.status.idle":"2021-09-27T14:22:13.032711Z","shell.execute_reply.started":"2021-09-27T14:22:12.959537Z","shell.execute_reply":"2021-09-27T14:22:13.031112Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Print out the graph to see the classification**","metadata":{}},{"cell_type":"code","source":"values = [train_data[train_data['target']==0].shape[0], train_data[train_data['target']==1].shape[0]]\nlabels = ['Sincere Questions', 'Insincere Questions']\n\nplt.pie(values, labels=labels, autopct='%1.1f%%', shadow=True)\nplt.title('Target Distribution')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-09-27T14:22:15.766556Z","iopub.execute_input":"2021-09-27T14:22:15.766866Z","iopub.status.idle":"2021-09-27T14:22:15.987808Z","shell.execute_reply.started":"2021-09-27T14:22:15.766836Z","shell.execute_reply":"2021-09-27T14:22:15.98676Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(num_sincere_questions/num_questions * 100, 'percent of training data questions are sincere')\nprint(num_insincere_questions/num_questions * 100, 'percent of training data questions are insincere')","metadata":{"execution":{"iopub.status.busy":"2021-09-27T14:22:17.350201Z","iopub.execute_input":"2021-09-27T14:22:17.350561Z","iopub.status.idle":"2021-09-27T14:22:17.356121Z","shell.execute_reply.started":"2021-09-27T14:22:17.350527Z","shell.execute_reply":"2021-09-27T14:22:17.355114Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2021-09-27T14:22:17.361313Z","iopub.execute_input":"2021-09-27T14:22:17.361916Z","iopub.status.idle":"2021-09-27T14:22:17.596118Z","shell.execute_reply.started":"2021-09-27T14:22:17.361886Z","shell.execute_reply":"2021-09-27T14:22:17.595262Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.duplicated(subset = [\"question_text\", \"qid\", \"target\"]).any()","metadata":{"execution":{"iopub.status.busy":"2021-09-27T14:22:17.597551Z","iopub.execute_input":"2021-09-27T14:22:17.597888Z","iopub.status.idle":"2021-09-27T14:22:18.441046Z","shell.execute_reply.started":"2021-09-27T14:22:17.597852Z","shell.execute_reply":"2021-09-27T14:22:18.439884Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Print out the info of the train data and test data**","metadata":{}},{"cell_type":"code","source":"train_data.info()","metadata":{"execution":{"iopub.status.busy":"2021-09-27T14:22:21.055917Z","iopub.execute_input":"2021-09-27T14:22:21.056239Z","iopub.status.idle":"2021-09-27T14:22:21.292446Z","shell.execute_reply.started":"2021-09-27T14:22:21.056209Z","shell.execute_reply":"2021-09-27T14:22:21.291126Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data.info()","metadata":{"execution":{"iopub.status.busy":"2021-09-27T14:22:21.951927Z","iopub.execute_input":"2021-09-27T14:22:21.952257Z","iopub.status.idle":"2021-09-27T14:22:22.027508Z","shell.execute_reply.started":"2021-09-27T14:22:21.952225Z","shell.execute_reply":"2021-09-27T14:22:22.026357Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Use Natural Language Tootkit to clean the data**\n+ Wordnet: It groups English words into sets of synonyms called synonym series, provides brief definitions and usage examples, and records the number of relationships between these synonym series or members <br>\n+ Punkt: Punkt Sentence Tokenizer. This tokenizer divides a text into a list of sentences, by using an unsupervised algorithm to build a model for abbreviation words, collocations, and words that start sentences <br>\n+ Stopwords: For the purpose of analyzing text data and building NLP models, these stopwords might not add much value to the meaning of the document --> We have to remove these words","metadata":{}},{"cell_type":"code","source":"nltk.download('wordnet')\nnltk.download('punkt')\nnltk.download('stopwords')","metadata":{"execution":{"iopub.status.busy":"2021-09-27T14:22:29.350252Z","iopub.execute_input":"2021-09-27T14:22:29.350624Z","iopub.status.idle":"2021-09-27T14:22:29.579286Z","shell.execute_reply.started":"2021-09-27T14:22:29.350591Z","shell.execute_reply":"2021-09-27T14:22:29.578324Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Process the raw data text to see the categories in the sentence**\n# **Print out some of the first data**","metadata":{}},{"cell_type":"code","source":"train_data['freq_qid'] = train_data.groupby('qid')['qid'].transform('count') \ntrain_data['qlen'] = train_data['question_text'].str.len() \ntrain_data['n_words'] = train_data['question_text'].apply(lambda row: len(row.split(\" \")))\ntrain_data['numeric_words'] = train_data['question_text'].apply(lambda row: sum(c.isdigit() for c in row))\ntrain_data['sp_char_words'] = train_data['question_text'].str.findall(r'[^a-zA-Z0–9 ]').str.len()\ntrain_data['char_words'] = train_data['question_text'].apply(lambda row: len(str(row)))\ntrain_data['unique_words'] = train_data['question_text'].apply(lambda row: len(set(str(row).split())))\n\ntrain_data.head()","metadata":{"execution":{"iopub.status.busy":"2021-09-27T14:22:34.034802Z","iopub.execute_input":"2021-09-27T14:22:34.03511Z","iopub.status.idle":"2021-09-27T14:22:56.225351Z","shell.execute_reply.started":"2021-09-27T14:22:34.035079Z","shell.execute_reply":"2021-09-27T14:22:56.224581Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data['freq_qid'] = test_data.groupby('qid')['qid'].transform('count') \ntest_data['qlen'] = test_data['question_text'].str.len() \ntest_data['n_words'] = test_data['question_text'].apply(lambda row: len(row.split(\" \")))\ntest_data['numeric_words'] = test_data['question_text'].apply(lambda row: sum(c.isdigit() for c in row))\ntest_data['sp_char_words'] = test_data['question_text'].str.findall(r'[^a-zA-Z0–9 ]').str.len()\ntest_data['char_words'] = test_data['question_text'].apply(lambda row: len(str(row)))\ntest_data['unique_words'] = test_data['question_text'].apply(lambda row: len(set(str(row).split())))\n\ntest_data.head()","metadata":{"execution":{"iopub.status.busy":"2021-09-27T14:22:56.226831Z","iopub.execute_input":"2021-09-27T14:22:56.227161Z","iopub.status.idle":"2021-09-27T14:23:02.248402Z","shell.execute_reply.started":"2021-09-27T14:22:56.227123Z","shell.execute_reply":"2021-09-27T14:23:02.247634Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Note: Toxic questions have average of more words and number than non-toxic questions**\n# **Insincere questions usually have bad meaning words rather than grammar**\n**Can use this feature into model (Tested but no good with linear models)**","metadata":{}},{"cell_type":"markdown","source":"---------------------------------------------------------------------------------------------","metadata":{}},{"cell_type":"markdown","source":"# **Special letters, numbers, paths, uppercase or lowercase usually do not affect the classification of the question, so they can be omitted**","metadata":{}},{"cell_type":"markdown","source":"****","metadata":{}},{"cell_type":"markdown","source":"**Remove special characters**","metadata":{}},{"cell_type":"code","source":"puncts=[',', '.', '\"', ':', ')', '(', '-', '!', '?', '|', ';', \"'\", '$', '&', '/', '[', ']', '>', '%', '=', '#', '*', '+', '\\\\', \n        '•', '~', '@', '£', '·', '_', '{', '}', '©', '^', '®', '`', '<', '→', '°', '€', '™', '›', '♥', '←', '×', '§', '″', '′', \n        '█', '…', '“', '★', '”', '–', '●', '►', '−', '¢', '¬', '░', '¡', '¶', '↑', '±', '¿', '▾', '═', '¦', '║', '―', '¥', '▓', \n        '—', '‹', '─', '▒', '：', '⊕', '▼', '▪', '†', '■', '’', '▀', '¨', '▄', '♫', '☆', '¯', '♦', '¤', '▲', '¸', '⋅', '‘', '∞', \n        '∙', '）', '↓', '、', '│', '（', '»', '，', '♪', '╩', '╚', '・', '╦', '╣', '╔', '╗', '▬', '❤', '≤', '‡', '√', '◄', '━', \n        '⇒', '▶', '≥', '╝', '♡', '◊', '。', '✈', '≡', '☺', '✔', '↵', '≈', '✓', '♣', '☎', '℃', '◦', '└', '‟', '～', '！', '○', \n        '◆', '№', '♠', '▌', '✿', '▸', '⁄', '□', '❖', '✦', '．', '÷', '｜', '┃', '／', '￥', '╠', '↩', '✭', '▐', '☼', '☻', '┐', \n        '├', '«', '∼', '┌', '℉', '☮', '฿', '≦', '♬', '✧', '〉', '－', '⌂', '✖', '･', '◕', '※', '‖', '◀', '‰', '\\x97', '↺', \n        '∆', '┘', '┬', '╬', '،', '⌘', '⊂', '＞', '〈', '⎙', '？', '☠', '⇐', '▫', '∗', '∈', '≠', '♀', '♔', '˚', '℗', '┗', '＊', \n        '┼', '❀', '＆', '∩', '♂', '‿', '∑', '‣', '➜', '┛', '⇓', '☯', '⊖', '☀', '┳', '；', '∇', '⇑', '✰', '◇', '♯', '☞', '´', \n        '↔', '┏', '｡', '◘', '∂', '✌', '♭', '┣', '┴', '┓', '✨', '\\xa0', '˜', '❥', '┫', '℠', '✒', '［', '∫', '\\x93', '≧', '］', \n        '\\x94', '∀', '♛', '\\x96', '∨', '◎', '↻', '⇩', '＜', '≫', '✩', '✪', '♕', '؟', '₤', '☛', '╮', '␊', '＋', '┈', '％', \n        '╋', '▽', '⇨', '┻', '⊗', '￡', '।', '▂', '✯', '▇', '＿', '➤', '✞', '＝', '▷', '△', '◙', '▅', '✝', '∧', '␉', '☭', \n        '┊', '╯', '☾', '➔', '∴', '\\x92', '▃', '↳', '＾', '׳', '➢', '╭', '➡', '＠', '⊙', '☢', '˝', '∏', '„', '∥', '❝', '☐', \n        '▆', '╱', '⋙', '๏', '☁', '⇔', '▔', '\\x91', '➚', '◡', '╰', '\\x85', '♢', '˙', '۞', '✘', '✮', '☑', '⋆', 'ⓘ', '❒', \n        '☣', '✉', '⌊', '➠', '∣', '❑', '◢', 'ⓒ', '\\x80', '〒', '∕', '▮', '⦿', '✫', '✚', '⋯', '♩', '☂', '❞', '‗', '܂', '☜', \n        '‾', '✜', '╲', '∘', '⟩', '＼', '⟨', '·', '✗', '♚', '∅', 'ⓔ', '◣', '͡', '‛', '❦', '◠', '✄', '❄', '∃', '␣', '≪', '｢', \n        '≅', '◯', '☽', '∎', '｣', '❧', '̅', 'ⓐ', '↘', '⚓', '▣', '˘', '∪', '⇢', '✍', '⊥', '＃', '⎯', '↠', '۩', '☰', '◥', \n        '⊆', '✽', '⚡', '↪', '❁', '☹', '◼', '☃', '◤', '❏', 'ⓢ', '⊱', '➝', '̣', '✡', '∠', '｀', '▴', '┤', '∝', '♏', 'ⓐ', \n        '✎', ';', '␤', '＇', '❣', '✂', '✤', 'ⓞ', '☪', '✴', '⌒', '˛', '♒', '＄', '✶', '▻', 'ⓔ', '◌', '◈', '❚', '❂', '￦', \n        '◉', '╜', '̃', '✱', '╖', '❉', 'ⓡ', '↗', 'ⓣ', '♻', '➽', '׀', '✲', '✬', '☉', '▉', '≒', '☥', '⌐', '♨', '✕', 'ⓝ', \n        '⊰', '❘', '＂', '⇧', '̵', '➪', '▁', '▏', '⊃', 'ⓛ', '‚', '♰', '́', '✏', '⏑', '̶', 'ⓢ', '⩾', '￠', '❍', '≃', '⋰', '♋', \n        '､', '̂', '❋', '✳', 'ⓤ', '╤', '▕', '⌣', '✸', '℮', '⁺', '▨', '╨', 'ⓥ', '♈', '❃', '☝', '✻', '⊇', '≻', '♘', '♞', \n        '◂', '✟', '⌠', '✠', '☚', '✥', '❊', 'ⓒ', '⌈', '❅', 'ⓡ', '♧', 'ⓞ', '▭', '❱', 'ⓣ', '∟', '☕', '♺', '∵', '⍝', 'ⓑ', \n        '✵', '✣', '٭', '♆', 'ⓘ', '∶', '⚜', '◞', '்', '✹', '➥', '↕', '̳', '∷', '✋', '➧', '∋', '̿', 'ͧ', '┅', '⥤', '⬆', '⋱', \n        '☄', '↖', '⋮', '۔', '♌', 'ⓛ', '╕', '♓', '❯', '♍', '▋', '✺', '⭐', '✾', '♊', '➣', '▿', 'ⓑ', '♉', '⏠', '◾', '▹', \n        '⩽', '↦', '╥', '⍵', '⌋', '։', '➨', '∮', '⇥', 'ⓗ', 'ⓓ', '⁻', '⎝', '⌥', '⌉', '◔', '◑', '✼', '♎', '♐', '╪', '⊚', \n        '☒', '⇤', 'ⓜ', '⎠', '◐', '⚠', '╞', '◗', '⎕', 'ⓨ', '☟', 'ⓟ', '♟', '❈', '↬', 'ⓓ', '◻', '♮', '❙', '♤', '∉', '؛', \n        '⁂', 'ⓝ', '־', '♑', '╫', '╓', '╳', '⬅', '☔', '☸', '┄', '╧', '׃', '⎢', '❆', '⋄', '⚫', '̏', '☏', '➞', '͂', '␙', \n        'ⓤ', '◟', '̊', '⚐', '✙', '↙', '̾', '℘', '✷', '⍺', '❌', '⊢', '▵', '✅', 'ⓖ', '☨', '▰', '╡', 'ⓜ', '☤', '∽', '╘', \n        '˹', '↨', '♙', '⬇', '♱', '⌡', '⠀', '╛', '❕', '┉', 'ⓟ', '̀', '♖', 'ⓚ', '┆', '⎜', '◜', '⚾', '⤴', '✇', '╟', '⎛', \n        '☩', '➲', '➟', 'ⓥ', 'ⓗ', '⏝', '◃', '╢', '↯', '✆', '˃', '⍴', '❇', '⚽', '╒', '̸', '♜', '☓', '➳', '⇄', '☬', '⚑', \n        '✐', '⌃', '◅', '▢', '❐', '∊', '☈', '॥', '⎮', '▩', 'ு', '⊹', '‵', '␔', '☊', '➸', '̌', '☿', '⇉', '⊳', '╙', 'ⓦ', \n        '⇣', '｛', '̄', '↝', '⎟', '▍', '❗', '״', '΄', '▞', '◁', '⛄', '⇝', '⎪', '♁', '⇠', '☇', '✊', 'ி', '｝', '⭕', '➘', \n        '⁀', '☙', '❛', '❓', '⟲', '⇀', '≲', 'ⓕ', '⎥', '\\u06dd', 'ͤ', '₋', '̱', '̎', '♝', '≳', '▙', '➭', '܀', 'ⓖ', '⇛', '▊', \n        '⇗', '̷', '⇱', '℅', 'ⓧ', '⚛', '̐', '̕', '⇌', '␀', '≌', 'ⓦ', '⊤', '̓', '☦', 'ⓕ', '▜', '➙', 'ⓨ', '⌨', '◮', '☷', \n        '◍', 'ⓚ', '≔', '⏩', '⍳', '℞', '┋', '˻', '▚', '≺', 'ْ', '▟', '➻', '̪', '⏪', '̉', '⎞', '┇', '⍟', '⇪', '▎', '⇦', '␝', \n        '⤷', '≖', '⟶', '♗', '̴', '♄', 'ͨ', '̈', '❜', '̡', '▛', '✁', '➩', 'ா', '˂', '↥', '⏎', '⎷', '̲', '➖', '↲', '⩵', '̗', '❢', \n        '≎', '⚔', '⇇', '̑', '⊿', '̖', '☍', '➹', '⥊', '⁁', '✢']\n\ndef clean_punct(x):\n    for punct in puncts:\n        if punct in x:\n            x = x.replace(punct, '{}' .format(punct))\n    return x","metadata":{"execution":{"iopub.status.busy":"2021-09-27T14:23:02.250282Z","iopub.execute_input":"2021-09-27T14:23:02.250635Z","iopub.status.idle":"2021-09-27T14:23:02.283906Z","shell.execute_reply.started":"2021-09-27T14:23:02.250598Z","shell.execute_reply":"2021-09-27T14:23:02.282956Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Remove number**","metadata":{}},{"cell_type":"code","source":"def clean_numbers(x):\n    if bool(re.search(r'\\d', x)):\n        x = re.sub('[0-9]{5,}', '#####', x)\n        x = re.sub('[0-9]{4}', '####', x)\n        x = re.sub('[0-9]{3}', '###', x)\n        x = re.sub('[0-9]{2}', '##', x)\n    return x","metadata":{"execution":{"iopub.status.busy":"2021-09-27T14:23:02.285479Z","iopub.execute_input":"2021-09-27T14:23:02.286054Z","iopub.status.idle":"2021-09-27T14:23:02.297121Z","shell.execute_reply.started":"2021-09-27T14:23:02.286016Z","shell.execute_reply":"2021-09-27T14:23:02.296375Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":" **Create a vector of mispell words** <br>\n **Convert the shortened form to the original**","metadata":{}},{"cell_type":"code","source":"mispell_dict = {'colour': 'color', 'centre': 'center', 'favourite': 'favorite', 'travelling': 'traveling', 'counselling': 'counseling', 'theatre': 'theater', 'cancelled': 'canceled', 'labour': 'labor', 'organisation': 'organization', 'wwii': 'world war 2', 'citicise': 'criticize', 'youtu ': 'youtube ', 'Qoura': 'Quora', 'sallary': 'salary', 'Whta': 'What', 'narcisist': 'narcissist', 'howdo': 'how do', 'whatare': 'what are', 'howcan': 'how can', 'howmuch': 'how much', 'howmany': 'how many', 'whydo': 'why do', 'doI': 'do I', 'theBest': 'the best', 'howdoes': 'how does', 'mastrubation': 'masturbation', 'mastrubate': 'masturbate', \"mastrubating\": 'masturbating', 'pennis': 'penis', 'Etherium': 'bitcoin', 'narcissit': 'narcissist', 'bigdata': 'big data', '2k17': '2017', '2k18': '2018', 'qouta': 'quota', 'exboyfriend': 'ex boyfriend', 'airhostess': 'air hostess', \"whst\": 'what', 'watsapp': 'whatsapp', 'demonitisation': 'demonetization', 'demonitization': 'demonetization', 'demonetisation': 'demonetization', \n                'electroneum':'bitcoin','nanodegree':'degree','hotstar':'star','dream11':'dream','ftre':'fire','tensorflow':'framework','unocoin':'bitcoin',\n                'lnmiit':'limit','unacademy':'academy','altcoin':'bitcoin','altcoins':'bitcoin','litecoin':'bitcoin','coinbase':'bitcoin','cryptocurency':'cryptocurrency',\n                'simpliv':'simple','quoras':'quora','schizoids':'psychopath','remainers':'remainder','twinflame':'soulmate','quorans':'quora','brexit':'demonetized',\n                'iiest':'institute','dceu':'comics','pessat':'exam','uceed':'college','bhakts':'devotee','boruto':'anime',\n                'cryptocoin':'bitcoin','blockchains':'blockchain','fiancee':'fiance','redmi':'smartphone','oneplus':'smartphone','qoura':'quora','deepmind':'framework','ryzen':'cpu','whattsapp':'whatsapp',\n                'undertale':'adventure','zenfone':'smartphone','cryptocurencies':'cryptocurrencies','koinex':'bitcoin','zebpay':'bitcoin','binance':'bitcoin','whtsapp':'whatsapp',\n                'reactjs':'framework','bittrex':'bitcoin','bitconnect':'bitcoin','bitfinex':'bitcoin','yourquote':'your quote','whyis':'why is','jiophone':'smartphone',\n                'dogecoin':'bitcoin','onecoin':'bitcoin','poloniex':'bitcoin','7700k':'cpu','angular2':'framework','segwit2x':'bitcoin','hashflare':'bitcoin','940mx':'gpu',\n                'openai':'framework','hashflare':'bitcoin','1050ti':'gpu','nearbuy':'near buy','freebitco':'bitcoin','antminer':'bitcoin','filecoin':'bitcoin','whatapp':'whatsapp',\n                'empowr':'empower','1080ti':'gpu','crytocurrency':'cryptocurrency','8700k':'cpu','whatsaap':'whatsapp','g4560':'cpu','payymoney':'pay money',\n                'fuckboys':'fuck boys','intenship':'internship','zcash':'bitcoin','demonatisation':'demonetization','narcicist':'narcissist','mastuburation':'masturbation',\n                'trignometric':'trigonometric','cryptocurreny':'cryptocurrency','howdid':'how did','crytocurrencies':'cryptocurrencies','phycopath':'psychopath',\n                'bytecoin':'bitcoin','possesiveness':'possessiveness','scollege':'college','humanties':'humanities','altacoin':'bitcoin','demonitised':'demonetized',\n                'brasília':'brazilia','accolite':'accolyte','econimics':'economics','varrier':'warrier','quroa':'quora','statergy':'strategy','langague':'language',\n                'splatoon':'game','7600k':'cpu','gate2018':'gate 2018','in2018':'in 2018','narcassist':'narcissist','jiocoin':'bitcoin','hnlu':'hulu','7300hq':'cpu',\n                'weatern':'western','interledger':'blockchain','deplation':'deflation', 'cryptocurrencies':'cryptocurrency', 'bitcoin':'blockchain cryptocurrency',}\n\ndef _get_mispell(mispell_dict):\n    mispell_re = re.compile('(%s)' % '|'.join(mispell_dict.keys()))\n    return mispell_dict, mispell_re\n\nmispellings, mispellings_re = _get_mispell(mispell_dict)\ndef replace_typical_misspell(text):\n    def replace(match):\n        return mispellings[match.group(0)]\n    return mispellings_re.sub(replace, text)","metadata":{"execution":{"iopub.status.busy":"2021-09-27T14:23:02.298508Z","iopub.execute_input":"2021-09-27T14:23:02.299083Z","iopub.status.idle":"2021-09-27T14:23:02.32093Z","shell.execute_reply.started":"2021-09-27T14:23:02.299046Z","shell.execute_reply":"2021-09-27T14:23:02.319905Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Convert abbreviated words**","metadata":{}},{"cell_type":"code","source":"contraction_dict = {\"ain't\": \"is not\", \"aren't\": \"are not\",\"can't\": \"cannot\", \"'cause\": \"because\", \"could've\": \"could have\", \"couldn't\": \"could not\", \"didn't\": \"did not\",  \"doesn't\": \"does not\", \"don't\": \"do not\", \"hadn't\": \"had not\", \"hasn't\": \"has not\", \"haven't\": \"have not\", \"he'd\": \"he would\",\"he'll\": \"he will\", \"he's\": \"he is\", \"how'd\": \"how did\", \"how'd'y\": \"how do you\", \"how'll\": \"how will\", \"how's\": \"how is\",  \"I'd\": \"I would\", \"I'd've\": \"I would have\", \"I'll\": \"I will\", \"I'll've\": \"I will have\",\"I'm\": \"I am\", \"I've\": \"I have\", \"i'd\": \"i would\", \"i'd've\": \"i would have\", \"i'll\": \"i will\",  \"i'll've\": \"i will have\",\"i'm\": \"i am\", \"i've\": \"i have\", \"isn't\": \"is not\", \"it'd\": \"it would\", \"it'd've\": \"it would have\", \"it'll\": \"it will\", \"it'll've\": \"it will have\",\"it's\": \"it is\", \"let's\": \"let us\", \"ma'am\": \"madam\", \"mayn't\": \"may not\", \"might've\": \"might have\",\"mightn't\": \"might not\",\"mightn't've\": \"might not have\", \"must've\": \"must have\", \"mustn't\": \"must not\", \"mustn't've\": \"must not have\", \"needn't\": \"need not\", \"needn't've\": \"need not have\",\"o'clock\": \"of the clock\", \"oughtn't\": \"ought not\", \"oughtn't've\": \"ought not have\", \"shan't\": \"shall not\", \"sha'n't\": \"shall not\", \"shan't've\": \"shall not have\", \"she'd\": \"she would\", \"she'd've\": \"she would have\", \"she'll\": \"she will\", \"she'll've\": \"she will have\", \"she's\": \"she is\", \"should've\": \"should have\", \"shouldn't\": \"should not\", \"shouldn't've\": \"should not have\", \"so've\": \"so have\",\"so's\": \"so as\", \"this's\": \"this is\",\"that'd\": \"that would\", \"that'd've\": \"that would have\", \"that's\": \"that is\", \"there'd\": \"there would\", \"there'd've\": \"there would have\", \"there's\": \"there is\", \"here's\": \"here is\",\"they'd\": \"they would\", \"they'd've\": \"they would have\", \"they'll\": \"they will\", \"they'll've\": \"they will have\", \"they're\": \"they are\", \"they've\": \"they have\", \"to've\": \"to have\", \"wasn't\": \"was not\", \"we'd\": \"we would\", \"we'd've\": \"we would have\", \"we'll\": \"we will\", \"we'll've\": \"we will have\", \"we're\": \"we are\", \"we've\": \"we have\", \"weren't\": \"were not\", \"what'll\": \"what will\", \"what'll've\": \"what will have\", \"what're\": \"what are\",  \"what's\": \"what is\", \"what've\": \"what have\", \"when's\": \"when is\", \"when've\": \"when have\", \"where'd\": \"where did\", \"where's\": \"where is\", \"where've\": \"where have\", \"who'll\": \"who will\", \"who'll've\": \"who will have\", \"who's\": \"who is\", \"who've\": \"who have\", \"why's\": \"why is\", \"why've\": \"why have\", \"will've\": \"will have\", \"won't\": \"will not\", \"won't've\": \"will not have\", \"would've\": \"would have\", \"wouldn't\": \"would not\", \"wouldn't've\": \"would not have\", \"y'all\": \"you all\", \"y'all'd\": \"you all would\",\"y'all'd've\": \"you all would have\",\"y'all're\": \"you all are\",\"y'all've\": \"you all have\",\"you'd\": \"you would\", \"you'd've\": \"you would have\", \"you'll\": \"you will\", \"you'll've\": \"you will have\", \"you're\": \"you are\", \"you've\": \"you have\"}\n\ndef _get_contractions(contraction_dict):\n    contraction_re = re.compile('(%s)' % '|'.join(contraction_dict.keys()))\n    return contraction_dict, contraction_re\n\ncontractions, contractions_re = _get_contractions(contraction_dict)\n\ndef replace_contractions(text):\n    def replace(match):\n        return contractions[match.group(0)]\n    return contractions_re.sub(replace, text)","metadata":{"execution":{"iopub.status.busy":"2021-09-27T14:23:05.432871Z","iopub.execute_input":"2021-09-27T14:23:05.433184Z","iopub.status.idle":"2021-09-27T14:23:05.450567Z","shell.execute_reply.started":"2021-09-27T14:23:05.433153Z","shell.execute_reply":"2021-09-27T14:23:05.449602Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **In order to process we must clean text**","metadata":{}},{"cell_type":"markdown","source":" **Remove stopwords**","metadata":{}},{"cell_type":"code","source":"stopword_list = nltk.corpus.stopwords.words('english')\ndef remove_stopwords(text, is_lower_case=True):\n    tokenizer = ToktokTokenizer()\n    tokens = tokenizer.tokenize(text)\n    tokens = [token.strip() for token in tokens]\n    if is_lower_case:\n        filtered_tokens = [token for token in tokens if token not in stopword_list]\n    else:\n        filtered_tokens = [token for token in tokens if token.lower() not in stopword_list]\n    filtered_text = ' '.join(filtered_tokens)\n    return filtered_text","metadata":{"execution":{"iopub.status.busy":"2021-09-27T14:23:29.282845Z","iopub.execute_input":"2021-09-27T14:23:29.283177Z","iopub.status.idle":"2021-09-27T14:23:29.307706Z","shell.execute_reply.started":"2021-09-27T14:23:29.283147Z","shell.execute_reply":"2021-09-27T14:23:29.306887Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":" **Convert words with the same variation of a word into a single word**","metadata":{}},{"cell_type":"code","source":"from nltk.stem import  SnowballStemmer\nfrom nltk.tokenize.toktok import ToktokTokenizer\ndef stem_text(text):\n    tokenizer = ToktokTokenizer()\n    stemmer = SnowballStemmer('english')\n    tokens = tokenizer.tokenize(text)\n    tokens = [token.strip() for token in tokens]\n    tokens = [stemmer.stem(token) for token in tokens]\n    return ' '.join(tokens)","metadata":{"execution":{"iopub.status.busy":"2021-09-27T14:23:29.942748Z","iopub.execute_input":"2021-09-27T14:23:29.943075Z","iopub.status.idle":"2021-09-27T14:23:29.94975Z","shell.execute_reply.started":"2021-09-27T14:23:29.943045Z","shell.execute_reply":"2021-09-27T14:23:29.94783Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from nltk.stem import WordNetLemmatizer\nfrom nltk.tokenize.toktok import ToktokTokenizer\nwordnet_lemmatizer = WordNetLemmatizer()\ndef lemma_text(text):\n    tokenizer = ToktokTokenizer()\n    tokens = tokenizer.tokenize(text)\n    tokens = [token.strip() for token in tokens]\n    tokens = [wordnet_lemmatizer.lemmatize(token) for token in tokens]\n    return ' '.join(tokens)","metadata":{"execution":{"iopub.status.busy":"2021-09-27T14:23:29.95562Z","iopub.execute_input":"2021-09-27T14:23:29.955919Z","iopub.status.idle":"2021-09-27T14:23:29.962405Z","shell.execute_reply.started":"2021-09-27T14:23:29.955891Z","shell.execute_reply":"2021-09-27T14:23:29.961415Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":" **Clean sentences by using all above features**","metadata":{}},{"cell_type":"code","source":"def clean_sentence(x):\n    x = x.lower()\n    x = clean_punct(x)\n    x = clean_numbers(x)\n    x = replace_typical_misspell(x)\n    x = remove_stopwords(x)\n    x = replace_contractions(x)\n    x = stem_text(x)\n    x = lemma_text(x)\n    x = x.replace(\"'\",\"\")\n    return x","metadata":{"execution":{"iopub.status.busy":"2021-09-27T14:23:32.43386Z","iopub.execute_input":"2021-09-27T14:23:32.434196Z","iopub.status.idle":"2021-09-27T14:23:32.440128Z","shell.execute_reply.started":"2021-09-27T14:23:32.434165Z","shell.execute_reply":"2021-09-27T14:23:32.438728Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Print out some sentences after cleaning**","metadata":{}},{"cell_type":"code","source":"train_data['preprocessed_question_text'] = train_data['question_text'].apply(lambda x: clean_sentence(x))","metadata":{"execution":{"iopub.status.busy":"2021-09-27T14:23:35.392549Z","iopub.execute_input":"2021-09-27T14:23:35.392867Z","iopub.status.idle":"2021-09-27T14:33:18.7469Z","shell.execute_reply.started":"2021-09-27T14:23:35.392836Z","shell.execute_reply":"2021-09-27T14:33:18.745886Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Print out some sentences of train data after cleaning**","metadata":{}},{"cell_type":"code","source":"train_data.preprocessed_question_text.head()","metadata":{"execution":{"iopub.status.busy":"2021-09-27T14:33:18.748307Z","iopub.execute_input":"2021-09-27T14:33:18.748793Z","iopub.status.idle":"2021-09-27T14:33:18.756389Z","shell.execute_reply.started":"2021-09-27T14:33:18.748755Z","shell.execute_reply":"2021-09-27T14:33:18.755426Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data['preprocessed_question_text'] = test_data['question_text'].apply(lambda x: clean_sentence(x))","metadata":{"execution":{"iopub.status.busy":"2021-09-27T14:33:18.758467Z","iopub.execute_input":"2021-09-27T14:33:18.758931Z","iopub.status.idle":"2021-09-27T14:36:05.245156Z","shell.execute_reply.started":"2021-09-27T14:33:18.758891Z","shell.execute_reply":"2021-09-27T14:36:05.244349Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Print out some sentences of test data after cleaning**","metadata":{}},{"cell_type":"code","source":"test_data.preprocessed_question_text.head()","metadata":{"execution":{"iopub.status.busy":"2021-09-27T14:36:05.2467Z","iopub.execute_input":"2021-09-27T14:36:05.247017Z","iopub.status.idle":"2021-09-27T14:36:05.255133Z","shell.execute_reply.started":"2021-09-27T14:36:05.246982Z","shell.execute_reply":"2021-09-27T14:36:05.254088Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **A tag cloud: A novelty visual representation of text data to visualize free form text**","metadata":{}},{"cell_type":"code","source":"def cloud(text, title, size = (10,7)):\n    # Processing Text\n    words_list = text.unique().tolist()\n    words = ' '.join(words_list)\n    \n    wordcloud = WordCloud(width=800, height=400,\n                          collocations=False\n                         ).generate(words)\n    \n    # Output Visualization\n    fig = plt.figure(figsize=size, dpi=80, facecolor='k',edgecolor='k')\n    plt.imshow(wordcloud,interpolation='bilinear')\n    plt.axis('off')\n    plt.title(title, fontsize=25,color='w')\n    plt.tight_layout(pad=0)\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2021-09-27T14:37:29.913995Z","iopub.execute_input":"2021-09-27T14:37:29.914354Z","iopub.status.idle":"2021-09-27T14:37:29.920768Z","shell.execute_reply.started":"2021-09-27T14:37:29.914301Z","shell.execute_reply":"2021-09-27T14:37:29.919643Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Print out the visualization of words which appear in sincere questions (train.csv)**","metadata":{}},{"cell_type":"code","source":"cloud(train_data[train_data['target']==0]['question_text'], 'Sincere Questions On question_text')","metadata":{"execution":{"iopub.status.busy":"2021-09-27T14:36:05.266257Z","iopub.execute_input":"2021-09-27T14:36:05.266795Z","iopub.status.idle":"2021-09-27T14:36:26.182089Z","shell.execute_reply.started":"2021-09-27T14:36:05.266759Z","shell.execute_reply":"2021-09-27T14:36:26.181308Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Print out the visualization of words which appear in sincere questions (train.csv) AFTER cleaning**","metadata":{}},{"cell_type":"code","source":"cloud(train_data[train_data['target']==0]['preprocessed_question_text'], 'Sincere Questions On preprocessed_question_text')","metadata":{"execution":{"iopub.status.busy":"2021-09-27T14:36:26.183148Z","iopub.execute_input":"2021-09-27T14:36:26.183491Z","iopub.status.idle":"2021-09-27T14:36:40.069621Z","shell.execute_reply.started":"2021-09-27T14:36:26.183457Z","shell.execute_reply":"2021-09-27T14:36:40.068773Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Print out the visualization of words which appear in insincere questions (train.csv)**","metadata":{}},{"cell_type":"code","source":"cloud(train_data[train_data['target']==1]['question_text'], 'Insincere Questions On question_text')","metadata":{"execution":{"iopub.status.busy":"2021-09-27T14:36:40.071646Z","iopub.execute_input":"2021-09-27T14:36:40.071999Z","iopub.status.idle":"2021-09-27T14:36:42.8105Z","shell.execute_reply.started":"2021-09-27T14:36:40.071962Z","shell.execute_reply":"2021-09-27T14:36:42.809708Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Print out the visualization of words which appear in insincere questions (train.csv) AFTER cleaning**","metadata":{}},{"cell_type":"code","source":"cloud(train_data[train_data['target']==1]['preprocessed_question_text'], 'Insincere Questions On preprocessed_question_text')","metadata":{"execution":{"iopub.status.busy":"2021-09-27T14:36:42.812029Z","iopub.execute_input":"2021-09-27T14:36:42.812426Z","iopub.status.idle":"2021-09-27T14:36:44.869158Z","shell.execute_reply.started":"2021-09-27T14:36:42.812389Z","shell.execute_reply":"2021-09-27T14:36:44.868242Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"-----------------------------------------------------------------------------------------------------","metadata":{}}]}