{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport re","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-07-31T04:51:36.652240Z","iopub.execute_input":"2021-07-31T04:51:36.652673Z","iopub.status.idle":"2021-07-31T04:51:36.657698Z","shell.execute_reply.started":"2021-07-31T04:51:36.652634Z","shell.execute_reply":"2021-07-31T04:51:36.656375Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Load Dataset and print some questions","metadata":{"_uuid":"6c0fc564965ba7e9a4e06e39427b0e24c83bd825"}},{"cell_type":"code","source":"train_df = pd.read_csv(\"../input/quora-insincere-questions-classification/train.csv\")\nX_train = train_df[\"question_text\"].fillna(\"dieter\").values\ntest_df = pd.read_csv(\"../input/quora-insincere-questions-classification/test.csv\")\nX_test = test_df[\"question_text\"].fillna(\"dieter\").values\ny = train_df[\"target\"]\n\ntext = train_df['question_text']\n\nfor row in text[:10]:\n    print(row)","metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","scrolled":true,"execution":{"iopub.status.busy":"2021-07-31T04:51:36.659708Z","iopub.execute_input":"2021-07-31T04:51:36.660156Z","iopub.status.idle":"2021-07-31T04:51:41.653830Z","shell.execute_reply.started":"2021-07-31T04:51:36.660108Z","shell.execute_reply":"2021-07-31T04:51:41.652808Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 1. Remove Numbers\n**Example:** Which is best powerbank for iPhone 7 in India? -> Which is best powerbank for iPhone  in India?","metadata":{"_uuid":"e2bb57f780def82431b92aa614b51ad3b24ec69f"}},{"cell_type":"code","source":"def removeNumbers(text):\n    \"\"\" Removes integers \"\"\"\n    text = ''.join([i for i in text if not i.isdigit()])         \n    return text\n\ntext_removeNumbers = pd.DataFrame(columns=['TextBefore', 'TextAfter', 'Changed'])\ntext_removeNumbers['TextBefore'] = text.copy()\n","metadata":{"_uuid":"59ffcfcdd5548dba6da994f58bc9088f3e10d4bc","execution":{"iopub.status.busy":"2021-07-31T04:51:41.656264Z","iopub.execute_input":"2021-07-31T04:51:41.656725Z","iopub.status.idle":"2021-07-31T04:51:41.958841Z","shell.execute_reply.started":"2021-07-31T04:51:41.656676Z","shell.execute_reply":"2021-07-31T04:51:41.957828Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for index, row in text_removeNumbers.iterrows():\n    row['TextAfter'] = removeNumbers(row['TextBefore'])","metadata":{"_uuid":"54fa3bb88566401e54c15e40d188b0cbb25b6b51","execution":{"iopub.status.busy":"2021-07-31T04:51:41.960905Z","iopub.execute_input":"2021-07-31T04:51:41.961352Z","iopub.status.idle":"2021-07-31T04:53:54.596915Z","shell.execute_reply.started":"2021-07-31T04:51:41.961302Z","shell.execute_reply":"2021-07-31T04:53:54.596026Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"text_removeNumbers['Changed'] = np.where(text_removeNumbers['TextBefore']==text_removeNumbers['TextAfter'], 'no', 'yes')\nprint(\"{} of {} ({:.4f}%) questions have been changed.\".format(len(text_removeNumbers[text_removeNumbers['Changed']=='yes']), len(text_removeNumbers), 100*len(text_removeNumbers[text_removeNumbers['Changed']=='yes'])/len(text_removeNumbers)))","metadata":{"_uuid":"e5048efc7bdd6bf8d9b75ea774a1804495dd5dfe","execution":{"iopub.status.busy":"2021-07-31T04:53:54.598377Z","iopub.execute_input":"2021-07-31T04:53:54.598911Z","iopub.status.idle":"2021-07-31T04:53:55.939123Z","shell.execute_reply.started":"2021-07-31T04:53:54.598859Z","shell.execute_reply":"2021-07-31T04:53:55.937773Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for index, row in text_removeNumbers[text_removeNumbers['Changed']=='yes'].head().iterrows():\n    print(row['TextBefore'],'->',row['TextAfter'])","metadata":{"_uuid":"9148d67930f06161856bb97a5214a344ddf6e392","execution":{"iopub.status.busy":"2021-07-31T04:53:55.940728Z","iopub.execute_input":"2021-07-31T04:53:55.941114Z","iopub.status.idle":"2021-07-31T04:53:56.198210Z","shell.execute_reply.started":"2021-07-31T04:53:55.941075Z","shell.execute_reply":"2021-07-31T04:53:56.197349Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 2. Replace Repetitions of Punctuation\nThis technique:\n - replaces repetitions of exlamation marks with the tag \"multiExclamation\"\n - replaces repetitions of question marks with the tag \"multiQuestion\"\n - replaces repetitions of stop marks with the tag \"multiStop\"\n \n **Example:** How do I overcome the fear of facing an interview? It's killing me inside..what should I do? -> How do I overcome the fear of facing an interview? It's killing me inside multiStop what should I do?","metadata":{"_uuid":"a14dcfd5ce1b6c3530750a2d8874c1b0c3fcf046","trusted":true}},{"cell_type":"code","source":"def replaceMultiExclamationMark(text):\n    \"\"\" Replaces repetitions of exlamation marks \"\"\"\n    text = re.sub(r\"(\\!)\\1+\", ' multiExclamation ', text)\n    return text\n\ndef replaceMultiQuestionMark(text):\n    \"\"\" Replaces repetitions of question marks \"\"\"\n    text = re.sub(r\"(\\?)\\1+\", ' multiQuestion ', text)\n    return text\n\ndef replaceMultiStopMark(text):\n    \"\"\" Replaces repetitions of stop marks \"\"\"\n    text = re.sub(r\"(\\.)\\1+\", ' multiStop ', text)\n    return text\n\ntext_replaceRepOfPunct = pd.DataFrame(columns=['TextBefore', 'TextAfter', 'Changed'])\ntext_replaceRepOfPunct['TextBefore'] = text.copy()","metadata":{"_uuid":"aa1fb76fd024e5ec6abe6aa289760ee4895473ef","execution":{"iopub.status.busy":"2021-07-31T04:53:56.200308Z","iopub.execute_input":"2021-07-31T04:53:56.200841Z","iopub.status.idle":"2021-07-31T04:53:56.513198Z","shell.execute_reply.started":"2021-07-31T04:53:56.200803Z","shell.execute_reply":"2021-07-31T04:53:56.512395Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for index, row in text_replaceRepOfPunct.iterrows():\n    row['TextAfter'] = replaceMultiExclamationMark(row['TextBefore'])\n    row['TextAfter'] = replaceMultiQuestionMark(row['TextBefore'])\n    row['TextAfter'] = replaceMultiStopMark(row['TextBefore'])","metadata":{"_uuid":"bd50b040f6580ff0df59c6aef4b2bed74fd5a323","execution":{"iopub.status.busy":"2021-07-31T04:53:56.514700Z","iopub.execute_input":"2021-07-31T04:53:56.515147Z","iopub.status.idle":"2021-07-31T04:56:42.378989Z","shell.execute_reply.started":"2021-07-31T04:53:56.515115Z","shell.execute_reply":"2021-07-31T04:56:42.377632Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"text_replaceRepOfPunct['Changed'] = np.where(text_replaceRepOfPunct['TextBefore']==text_replaceRepOfPunct['TextAfter'], 'no', 'yes')\nprint(\"{} of {} ({:.4f}%) questions have been changed.\".format(len(text_replaceRepOfPunct[text_replaceRepOfPunct['Changed']=='yes']), len(text_replaceRepOfPunct), 100*len(text_replaceRepOfPunct[text_replaceRepOfPunct['Changed']=='yes'])/len(text_replaceRepOfPunct)))","metadata":{"scrolled":true,"_uuid":"4d8281d71d95fa7a299593122fdb074863110547","execution":{"iopub.status.busy":"2021-07-31T04:56:42.380623Z","iopub.execute_input":"2021-07-31T04:56:42.380993Z","iopub.status.idle":"2021-07-31T04:56:43.535798Z","shell.execute_reply.started":"2021-07-31T04:56:42.380955Z","shell.execute_reply":"2021-07-31T04:56:43.534698Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for index, row in text_replaceRepOfPunct[text_replaceRepOfPunct['Changed']=='yes'].head().iterrows():\n    print(row['TextBefore'],'->',row['TextAfter'])","metadata":{"_uuid":"9f8305a96b11d645a3ad024b265f00ff4ebdad39","execution":{"iopub.status.busy":"2021-07-31T04:56:43.537499Z","iopub.execute_input":"2021-07-31T04:56:43.538152Z","iopub.status.idle":"2021-07-31T04:56:43.758472Z","shell.execute_reply.started":"2021-07-31T04:56:43.538097Z","shell.execute_reply":"2021-07-31T04:56:43.757333Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 3. Remove Punctuation\n**Example:** Why haven't two democracies never ever went for a full fledged war? What stops them? -> Why havent two democracies never ever went for a full fledged war What stops them","metadata":{"_uuid":"bc911c610cc2d7cb845f36c4c1cad7591317c0cc"}},{"cell_type":"code","source":"import string\ntranslator = str.maketrans('', '', string.punctuation)\ntext_removePunctuation = pd.DataFrame(columns=['TextBefore', 'TextAfter', 'Changed'])\ntext_removePunctuation['TextBefore'] = text.copy()","metadata":{"_uuid":"e8db070a372cc0c04fada1ab95bf01d456abcc24","execution":{"iopub.status.busy":"2021-07-31T04:56:43.761725Z","iopub.execute_input":"2021-07-31T04:56:43.762079Z","iopub.status.idle":"2021-07-31T04:56:44.076976Z","shell.execute_reply.started":"2021-07-31T04:56:43.762044Z","shell.execute_reply":"2021-07-31T04:56:44.075752Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for index, row in text_removePunctuation.iterrows():\n    row['TextAfter'] = row['TextBefore'].translate(translator) ","metadata":{"_uuid":"70083f413bf6411d7fedb7d7a799dba6efbc03f3","execution":{"iopub.status.busy":"2021-07-31T04:56:44.078597Z","iopub.execute_input":"2021-07-31T04:56:44.079098Z","iopub.status.idle":"2021-07-31T04:58:48.514675Z","shell.execute_reply.started":"2021-07-31T04:56:44.079051Z","shell.execute_reply":"2021-07-31T04:58:48.513468Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"text_removePunctuation['Changed'] = np.where(text_removePunctuation['TextBefore']==text_removePunctuation['TextAfter'], 'no', 'yes')\nprint(\"{} of {} ({:.4f}%) questions have been changed.\".format(len(text_removePunctuation[text_removePunctuation['Changed']=='yes']), len(text_removePunctuation), 100*len(text_removePunctuation[text_removePunctuation['Changed']=='yes'])/len(text_removePunctuation)))","metadata":{"_uuid":"9c96b84ffddd998bc73f641538abab7e498f1ac3","execution":{"iopub.status.busy":"2021-07-31T04:58:48.516152Z","iopub.execute_input":"2021-07-31T04:58:48.516518Z","iopub.status.idle":"2021-07-31T04:58:50.076079Z","shell.execute_reply.started":"2021-07-31T04:58:48.516483Z","shell.execute_reply":"2021-07-31T04:58:50.074995Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for index, row in text_removePunctuation[text_removePunctuation['Changed']=='yes'].head().iterrows():\n    print(row['TextBefore'],'->',row['TextAfter'])","metadata":{"scrolled":true,"_uuid":"0fc301c2cc218739bb603f2b22ddaebb483ef54c","execution":{"iopub.status.busy":"2021-07-31T04:58:50.077764Z","iopub.execute_input":"2021-07-31T04:58:50.078142Z","iopub.status.idle":"2021-07-31T04:58:50.432499Z","shell.execute_reply.started":"2021-07-31T04:58:50.078105Z","shell.execute_reply":"2021-07-31T04:58:50.431190Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"I expected everything to change, because they are question with \"?\". Let's see the ones that didn't change.","metadata":{"_uuid":"3545ca50d76dbfe5a99c6806b51c60f70979249a"}},{"cell_type":"code","source":"for index, row in text_removePunctuation[text_removePunctuation['Changed']=='no'].head().iterrows():\n    print(row['TextBefore'],'->',row['TextAfter'])","metadata":{"_uuid":"818a2187bb542210bf93e9467d3ee33319862719","execution":{"iopub.status.busy":"2021-07-31T04:58:50.434128Z","iopub.execute_input":"2021-07-31T04:58:50.434547Z","iopub.status.idle":"2021-07-31T04:58:50.675189Z","shell.execute_reply.started":"2021-07-31T04:58:50.434508Z","shell.execute_reply":"2021-07-31T04:58:50.674040Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 4. Replace Contractions\nThis techniques replaces contractions to their equivalents.\n\n**Example:** What's the scariest thing that ever happened to anyone? -> What is the scariest thing that ever happened to anyone?","metadata":{"_uuid":"cf9fcef515a0e716a6a3c6e6bc44cb4928cd54c1"}},{"cell_type":"code","source":"contraction_patterns = [ (r'won\\'t', 'will not'), (r'can\\'t', 'cannot'), (r'i\\'m', 'i am'), (r'ain\\'t', 'is not'), (r'(\\w+)\\'ll', '\\g<1> will'), (r'(\\w+)n\\'t', '\\g<1> not'),\n                         (r'(\\w+)\\'ve', '\\g<1> have'), (r'(\\w+)\\'s', '\\g<1> is'), (r'(\\w+)\\'re', '\\g<1> are'), (r'(\\w+)\\'d', '\\g<1> would'), (r'&', 'and'), (r'dammit', 'damn it'), (r'dont', 'do not'), (r'wont', 'will not') ]\ndef replaceContraction(text):\n    patterns = [(re.compile(regex), repl) for (regex, repl) in contraction_patterns]\n    for (pattern, repl) in patterns:\n        (text, count) = re.subn(pattern, repl, text)\n    return text\n\ntext_replaceContractions = pd.DataFrame(columns=['TextBefore', 'TextAfter', 'Changed'])\ntext_replaceContractions['TextBefore'] = text.copy()","metadata":{"_uuid":"47a8feccbc1364a3c0aa5a35a99e91f233927baa","execution":{"iopub.status.busy":"2021-07-31T04:58:50.676875Z","iopub.execute_input":"2021-07-31T04:58:50.677231Z","iopub.status.idle":"2021-07-31T04:58:50.995690Z","shell.execute_reply.started":"2021-07-31T04:58:50.677196Z","shell.execute_reply":"2021-07-31T04:58:50.994719Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for index, row in text_replaceContractions.iterrows():\n    row['TextAfter'] = replaceContraction(row['TextBefore'])","metadata":{"_uuid":"fa9db144bd7fa2dcd793225890a3d90c7ea41175","execution":{"iopub.status.busy":"2021-07-31T04:58:50.997555Z","iopub.execute_input":"2021-07-31T04:58:50.998033Z","iopub.status.idle":"2021-07-31T05:04:05.741041Z","shell.execute_reply.started":"2021-07-31T04:58:50.997983Z","shell.execute_reply":"2021-07-31T05:04:05.739862Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"text_replaceContractions['Changed'] = np.where(text_replaceContractions['TextBefore']==text_replaceContractions['TextAfter'], 'no', 'yes')\nprint(\"{} of {} ({:.4f}%) questions have been changed.\".format(len(text_replaceContractions[text_replaceContractions['Changed']=='yes']), len(text_replaceContractions), 100*len(text_replaceContractions[text_replaceContractions['Changed']=='yes'])/len(text_replaceContractions)))","metadata":{"_uuid":"7c90b57020aadc67f0911fe31ab6c53da330be78","execution":{"iopub.status.busy":"2021-07-31T05:04:05.742848Z","iopub.execute_input":"2021-07-31T05:04:05.743311Z","iopub.status.idle":"2021-07-31T05:04:06.988008Z","shell.execute_reply.started":"2021-07-31T05:04:05.743255Z","shell.execute_reply":"2021-07-31T05:04:06.986795Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for index, row in text_replaceContractions[text_replaceContractions['Changed']=='yes'].head().iterrows():\n    print(row['TextBefore'],'->',row['TextAfter'])","metadata":{"_uuid":"7b9af939d9e04c2b73d04f645cbfa37e2f48f343","execution":{"iopub.status.busy":"2021-07-31T05:04:06.993652Z","iopub.execute_input":"2021-07-31T05:04:06.994015Z","iopub.status.idle":"2021-07-31T05:04:07.240429Z","shell.execute_reply.started":"2021-07-31T05:04:06.993983Z","shell.execute_reply":"2021-07-31T05:04:07.239347Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 5. Lowercase\n**Example:** What do you know about Bram Fischer and the Rivonia Trial? -> what do you know about bram fischer and the rivonia trial?","metadata":{"_uuid":"475e14c586dd3c18a95a3e2318869bf847033632"}},{"cell_type":"code","source":"text_lowercase = pd.DataFrame(columns=['TextBefore', 'TextAfter', 'Changed'])\ntext_lowercase['TextBefore'] = text.copy()","metadata":{"_uuid":"355e6fc585fd9d49fc37219c14ab14004ffbb6ce","execution":{"iopub.status.busy":"2021-07-31T05:04:07.244683Z","iopub.execute_input":"2021-07-31T05:04:07.245056Z","iopub.status.idle":"2021-07-31T05:04:07.552645Z","shell.execute_reply.started":"2021-07-31T05:04:07.245016Z","shell.execute_reply":"2021-07-31T05:04:07.551531Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for index, row in text_lowercase.iterrows():\n    row['TextAfter'] = row['TextBefore'].lower()","metadata":{"_uuid":"5b39fd5e7afa0a476d08251151a0fdaf3bfe18f5","execution":{"iopub.status.busy":"2021-07-31T05:04:07.554009Z","iopub.execute_input":"2021-07-31T05:04:07.554332Z","iopub.status.idle":"2021-07-31T05:06:07.296141Z","shell.execute_reply.started":"2021-07-31T05:04:07.554297Z","shell.execute_reply":"2021-07-31T05:06:07.294902Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"text_lowercase['Changed'] = np.where(text_lowercase['TextBefore']==text_lowercase['TextAfter'], 'no', 'yes')\nprint(\"{} of {} ({:.4f}%) questions have been changed.\".format(len(text_lowercase[text_lowercase['Changed']=='yes']), len(text_lowercase), 100*len(text_lowercase[text_lowercase['Changed']=='yes'])/len(text_lowercase)))","metadata":{"_uuid":"7e9926f52d87ece340dee73deeb4d3ceaa11000e","execution":{"iopub.status.busy":"2021-07-31T05:06:07.297448Z","iopub.execute_input":"2021-07-31T05:06:07.297787Z","iopub.status.idle":"2021-07-31T05:06:08.821658Z","shell.execute_reply.started":"2021-07-31T05:06:07.297754Z","shell.execute_reply":"2021-07-31T05:06:08.820425Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for index, row in text_lowercase[text_lowercase['Changed']=='yes'].head().iterrows():\n    print(row['TextBefore'],'->',row['TextAfter'])","metadata":{"scrolled":true,"_uuid":"183b703c8cfe1ea0fa7393158a1fbcebf625f516","execution":{"iopub.status.busy":"2021-07-31T05:06:08.823086Z","iopub.execute_input":"2021-07-31T05:06:08.823422Z","iopub.status.idle":"2021-07-31T05:06:09.171280Z","shell.execute_reply.started":"2021-07-31T05:06:08.823387Z","shell.execute_reply":"2021-07-31T05:06:09.170240Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Some question are written only in lowercase. This happens when they start with a number.","metadata":{"_uuid":"ade19cce80271933534a3c1b9089dbc3bb757c34"}},{"cell_type":"code","source":"for index, row in text_lowercase[text_lowercase['Changed']=='no'].head().iterrows():\n    print(row['TextBefore'],'->',row['TextAfter'])","metadata":{"_uuid":"4aa2466287972dc11924194d281e3c856def51a8","execution":{"iopub.status.busy":"2021-07-31T05:06:09.172762Z","iopub.execute_input":"2021-07-31T05:06:09.173072Z","iopub.status.idle":"2021-07-31T05:06:09.409669Z","shell.execute_reply.started":"2021-07-31T05:06:09.173040Z","shell.execute_reply":"2021-07-31T05:06:09.408845Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 6. Replace Negations with Antonyms\n**Example:** Why are humans not able to be evolved developing resistance against diseases? -> Why are humans unable to be evolved developing resistance against diseases ?","metadata":{"_uuid":"a0e1e0caefab5a4ef24519dde8bb41d966fa1ffe","trusted":true}},{"cell_type":"code","source":"import nltk\nfrom nltk.corpus import wordnet\n\ndef replace(word, pos=None):\n    \"\"\" Creates a set of all antonyms for the word and if there is only one antonym, it returns it \"\"\"\n    antonyms = set()\n    for syn in wordnet.synsets(word, pos=pos):\n        for lemma in syn.lemmas():\n            for antonym in lemma.antonyms():\n                antonyms.add(antonym.name())\n    if len(antonyms) == 1:\n        return antonyms.pop()\n    else:\n        return None\n\ndef replaceNegations(text):\n    \"\"\" Finds \"not\" and antonym for the next word and if found, replaces not and the next word with the antonym \"\"\"\n    i, l = 0, len(text)\n    words = []\n    while i < l:\n        word = text[i]\n        if word == 'not' and i+1 < l:\n            ant = replace(text[i+1])\n            if ant:\n                words.append(ant)\n                i += 2\n                continue\n        words.append(word)\n        i += 1\n    return words\n\ndef tokenize1(text):\n    tokens = nltk.word_tokenize(text)\n    tokens = replaceNegations(tokens)\n    text = \" \".join(tokens)\n    return text\n\ntext_replaceNegations = pd.DataFrame(columns=['TextBefore', 'TextAfter', 'Changed'])\ntext_replaceNegations['TextBefore'] = text.copy()","metadata":{"_uuid":"af37ca08e1a1ec777c27b1b2c84ce6afb8b2efd4","execution":{"iopub.status.busy":"2021-07-31T05:06:09.410643Z","iopub.execute_input":"2021-07-31T05:06:09.410907Z","iopub.status.idle":"2021-07-31T05:06:11.293804Z","shell.execute_reply.started":"2021-07-31T05:06:09.410880Z","shell.execute_reply":"2021-07-31T05:06:11.292615Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for index, row in text_replaceNegations.iterrows():\n    row['TextAfter'] = tokenize1(row['TextBefore'])","metadata":{"_uuid":"479f4b3a4186f024059abed8f26ec8b309df17cf","execution":{"iopub.status.busy":"2021-07-31T05:06:11.295221Z","iopub.execute_input":"2021-07-31T05:06:11.295550Z","iopub.status.idle":"2021-07-31T05:14:15.507546Z","shell.execute_reply.started":"2021-07-31T05:06:11.295517Z","shell.execute_reply":"2021-07-31T05:14:15.505540Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"text_replaceNegations['Changed'] = np.where(text_replaceNegations['TextBefore'].str.replace(\" \",\"\")==text_replaceNegations['TextAfter'].str.replace(\" \",\"\").str.replace(\"``\",'\"').str.replace(\"''\",'\"'), 'no', 'yes')\nprint(\"{} of {} ({:.4f}%) questions have been changed.\".format(len(text_replaceNegations[text_replaceNegations['Changed']=='yes']), len(text_replaceNegations), 100*len(text_replaceNegations[text_replaceNegations['Changed']=='yes'])/len(text_replaceNegations)))","metadata":{"_uuid":"711021a1b0437d1a16e9a634fdd45bc115751bd4","execution":{"iopub.status.busy":"2021-07-31T05:14:15.509120Z","iopub.execute_input":"2021-07-31T05:14:15.509495Z","iopub.status.idle":"2021-07-31T05:14:23.439427Z","shell.execute_reply.started":"2021-07-31T05:14:15.509461Z","shell.execute_reply":"2021-07-31T05:14:23.438249Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for index, row in text_replaceNegations[text_replaceNegations['Changed']=='yes'].head().iterrows():\n    print(row['TextBefore'],'->',row['TextAfter'])","metadata":{"scrolled":true,"_uuid":"7ece918f2d15be189493537c2bece44ef76b5f83","execution":{"iopub.status.busy":"2021-07-31T05:14:23.441869Z","iopub.execute_input":"2021-07-31T05:14:23.442406Z","iopub.status.idle":"2021-07-31T05:14:23.639430Z","shell.execute_reply.started":"2021-07-31T05:14:23.442350Z","shell.execute_reply":"2021-07-31T05:14:23.638285Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 7. Handle Capitalized Words\n**Example:** Which is better to use, Avro or ORC? -> Which is better to use , Avro or ALL_CAPS_ORC ?","metadata":{"_uuid":"4bb1384b130c3a75099dac3fd8700449918c98ca"}},{"cell_type":"code","source":"def addCapTag(word):\n    \"\"\" Finds a word with at least 3 characters capitalized and adds the tag ALL_CAPS_ \"\"\"\n    if(len(re.findall(\"[A-Z]{3,}\", word))):\n        word = word.replace('\\\\', '' )\n        transformed = re.sub(\"[A-Z]{3,}\", \"ALL_CAPS_\"+word, word)\n        return transformed\n    else:\n        return word\n\ndef tokenize2(text):\n    finalTokens = []\n    tokens = nltk.word_tokenize(text)\n    for w in tokens:\n        finalTokens.append(addCapTag(w))\n    text = \" \".join(finalTokens)\n    return text\n\ntext_handleCapWords = pd.DataFrame(columns=['TextBefore', 'TextAfter', 'Changed'])\ntext_handleCapWords['TextBefore'] = text.copy()","metadata":{"_uuid":"12772ee36dbd66e7ec507959990516590e1f38e6","execution":{"iopub.status.busy":"2021-07-31T05:14:23.640921Z","iopub.execute_input":"2021-07-31T05:14:23.641402Z","iopub.status.idle":"2021-07-31T05:14:23.942956Z","shell.execute_reply.started":"2021-07-31T05:14:23.641361Z","shell.execute_reply":"2021-07-31T05:14:23.941757Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for index, row in text_handleCapWords.iterrows():\n    row['TextAfter'] = tokenize2(row['TextBefore'])","metadata":{"_uuid":"16e555ff71640676cb33f833fa289f72ec6991db","execution":{"iopub.status.busy":"2021-07-31T05:14:23.944564Z","iopub.execute_input":"2021-07-31T05:14:23.944959Z","iopub.status.idle":"2021-07-31T05:22:47.462377Z","shell.execute_reply.started":"2021-07-31T05:14:23.944921Z","shell.execute_reply":"2021-07-31T05:22:47.461223Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"text_handleCapWords['Changed'] = np.where(text_handleCapWords['TextBefore'].str.replace(\" \",\"\")==text_handleCapWords['TextAfter'].str.replace(\" \",\"\").str.replace(\"``\",'\"').str.replace(\"''\",'\"'), 'no', 'yes')\nprint(\"{} of {} ({:.4f}%) questions have been changed.\".format(len(text_handleCapWords[text_handleCapWords['Changed']=='yes']), len(text_handleCapWords), 100*len(text_handleCapWords[text_handleCapWords['Changed']=='yes'])/len(text_handleCapWords)))","metadata":{"_uuid":"5a732d04fe19e20103c6bca0d1ccd91245535eeb","execution":{"iopub.status.busy":"2021-07-31T05:22:47.464047Z","iopub.execute_input":"2021-07-31T05:22:47.464507Z","iopub.status.idle":"2021-07-31T05:22:55.312447Z","shell.execute_reply.started":"2021-07-31T05:22:47.464457Z","shell.execute_reply":"2021-07-31T05:22:55.311627Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for index, row in text_handleCapWords[text_handleCapWords['Changed']=='yes'].head().iterrows():\n    print(row['TextBefore'],'->',row['TextAfter'])","metadata":{"_uuid":"82e1c29c084ad1cfb0235b9b294069c9043038c5","execution":{"iopub.status.busy":"2021-07-31T05:22:55.313655Z","iopub.execute_input":"2021-07-31T05:22:55.314102Z","iopub.status.idle":"2021-07-31T05:22:55.532765Z","shell.execute_reply.started":"2021-07-31T05:22:55.314067Z","shell.execute_reply":"2021-07-31T05:22:55.531908Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 8. Remove Stopwords\n**Example:** The movie was not good at all. -> movie good","metadata":{"_uuid":"0fd322ed368820947deacbef2b5f9e6dce4042e4"}},{"cell_type":"code","source":"from nltk.corpus import stopwords\nstoplist = stopwords.words('english')\n\ndef tokenize(text):\n    finalTokens = []\n    tokens = nltk.word_tokenize(text)\n    for w in tokens:\n        if (w not in stoplist):\n            finalTokens.append(w)\n    text = \" \".join(finalTokens)\n    return text\n\ntext_removeStopwords = pd.DataFrame(columns=['TextBefore', 'TextAfter', 'Changed'])\ntext_removeStopwords['TextBefore'] = text.copy()","metadata":{"_uuid":"3a7db86307c657c34c85310aa80b44bc1989c427","execution":{"iopub.status.busy":"2021-07-31T05:22:55.534349Z","iopub.execute_input":"2021-07-31T05:22:55.534865Z","iopub.status.idle":"2021-07-31T05:22:55.829685Z","shell.execute_reply.started":"2021-07-31T05:22:55.534792Z","shell.execute_reply":"2021-07-31T05:22:55.828633Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for index, row in text_removeStopwords.iterrows():\n    row['TextAfter'] = tokenize(row['TextBefore'])","metadata":{"_uuid":"0a6fe3f5a52dff0d3ab120674583a669d3ed712b","execution":{"iopub.status.busy":"2021-07-31T05:22:55.831107Z","iopub.execute_input":"2021-07-31T05:22:55.831496Z","iopub.status.idle":"2021-07-31T05:31:17.580401Z","shell.execute_reply.started":"2021-07-31T05:22:55.831443Z","shell.execute_reply":"2021-07-31T05:31:17.579204Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"text_removeStopwords['Changed'] = np.where(text_removeStopwords['TextBefore'].str.replace(\" \",\"\")==text_removeStopwords['TextAfter'].str.replace(\" \",\"\").str.replace(\"``\",'\"').str.replace(\"''\",'\"'), 'no', 'yes')\nprint(\"{} of {} ({:.4f}%) questions have been changed.\".format(len(text_removeStopwords[text_removeStopwords['Changed']=='yes']), len(text_removeStopwords), 100*len(text_removeStopwords[text_removeStopwords['Changed']=='yes'])/len(text_removeStopwords)))","metadata":{"_uuid":"8e1482892be2e8b5427639d9fc81cf04d07e66ed","execution":{"iopub.status.busy":"2021-07-31T05:31:17.582083Z","iopub.execute_input":"2021-07-31T05:31:17.582475Z","iopub.status.idle":"2021-07-31T05:31:25.140201Z","shell.execute_reply.started":"2021-07-31T05:31:17.582440Z","shell.execute_reply":"2021-07-31T05:31:25.139022Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for index, row in text_removeStopwords[text_removeStopwords['Changed']=='yes'].head().iterrows():\n    print(row['TextBefore'],'->',row['TextAfter'])","metadata":{"_uuid":"c88343608802d3734eaf00998a1200992006e607","execution":{"iopub.status.busy":"2021-07-31T05:31:25.141575Z","iopub.execute_input":"2021-07-31T05:31:25.141946Z","iopub.status.idle":"2021-07-31T05:31:25.433868Z","shell.execute_reply.started":"2021-07-31T05:31:25.141912Z","shell.execute_reply":"2021-07-31T05:31:25.432549Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 9. Replace Elongated Words\nThis technique replaces an elongated word with its basic form, unless the word exists in the lexicon.\n\n**Example:** Game of Thrones, what does Arya find out about Littlefinger? -> Game of Thrones , what does Arya find out about Litlefinger ?","metadata":{"_uuid":"25b684706946cec462e5aec250125004eba65dce"}},{"cell_type":"code","source":"def replaceElongated(word):\n    \"\"\" Replaces an elongated word with its basic form, unless the word exists in the lexicon \"\"\"\n\n    repeat_regexp = re.compile(r'(\\w*)(\\w)\\2(\\w*)')\n    repl = r'\\1\\2\\3'\n    if wordnet.synsets(word):\n        return word\n    repl_word = repeat_regexp.sub(repl, word)\n    if repl_word != word:      \n        return replaceElongated(repl_word)\n    else:       \n        return repl_word\n    \ndef tokenize(text):\n    finalTokens = []\n    tokens = nltk.word_tokenize(text)\n    for w in tokens:\n        finalTokens.append(replaceElongated(w))\n    text = \" \".join(finalTokens)\n    return text\n\ntext_removeElWords = pd.DataFrame(columns=['TextBefore', 'TextAfter', 'Changed'])\ntext_removeElWords['TextBefore'] = text.copy()","metadata":{"_uuid":"8be40c531d37d808167b2263260774d43479035a","execution":{"iopub.status.busy":"2021-07-31T05:31:25.435553Z","iopub.execute_input":"2021-07-31T05:31:25.436020Z","iopub.status.idle":"2021-07-31T05:31:25.877561Z","shell.execute_reply.started":"2021-07-31T05:31:25.435967Z","shell.execute_reply":"2021-07-31T05:31:25.876512Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for index, row in text_removeElWords.iterrows():\n    row['TextAfter'] = tokenize(row['TextBefore'])","metadata":{"_uuid":"3e6ed9fe42bbb3db2fd908fd75f3f54a68d94176","execution":{"iopub.status.busy":"2021-07-31T05:31:25.879077Z","iopub.execute_input":"2021-07-31T05:31:25.879419Z","iopub.status.idle":"2021-07-31T05:47:38.535208Z","shell.execute_reply.started":"2021-07-31T05:31:25.879385Z","shell.execute_reply":"2021-07-31T05:47:38.534045Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"text_removeElWords['Changed'] = np.where(text_removeElWords['TextBefore'].str.replace(\" \",\"\")==text_removeElWords['TextAfter'].str.replace(\" \",\"\").str.replace(\"``\",'\"').str.replace(\"''\",'\"'), 'no', 'yes')\nprint(\"{} of {} ({:.4f}%) questions have been changed.\".format(len(text_removeElWords[text_removeElWords['Changed']=='yes']), len(text_removeElWords), 100*len(text_removeElWords[text_removeElWords['Changed']=='yes'])/len(text_removeElWords)))","metadata":{"_uuid":"b815a11ab55e4533f0e3c0642630c35c309feb72","execution":{"iopub.status.busy":"2021-07-31T05:47:38.537562Z","iopub.execute_input":"2021-07-31T05:47:38.537903Z","iopub.status.idle":"2021-07-31T05:47:46.301079Z","shell.execute_reply.started":"2021-07-31T05:47:38.537871Z","shell.execute_reply":"2021-07-31T05:47:46.300033Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for index, row in text_removeElWords[text_removeElWords['Changed']=='yes'].head().iterrows():\n    print(row['TextBefore'],'->',row['TextAfter'])","metadata":{"_uuid":"dcc2d2413e8b98ca26bbff3fd84c677a427ffecf","execution":{"iopub.status.busy":"2021-07-31T05:47:46.302517Z","iopub.execute_input":"2021-07-31T05:47:46.302823Z","iopub.status.idle":"2021-07-31T05:47:46.520234Z","shell.execute_reply.started":"2021-07-31T05:47:46.302795Z","shell.execute_reply":"2021-07-31T05:47:46.518884Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 10. Stemming/Lemmatizing\n**Example:** How do modern military submarines reduce noise to achieve stealth? -> how do modern militari submarin reduc nois to achiev stealth ?","metadata":{"_uuid":"9c478fb8e3810e1aa6385ae17cf026027e4d5f06"}},{"cell_type":"code","source":"from nltk.stem.porter import PorterStemmer\nstemmer = PorterStemmer() #set stemmer\nfrom nltk.stem import WordNetLemmatizer\nlemmatizer = WordNetLemmatizer() # set lemmatizer\n\ndef tokenize(text):\n    finalTokens = []\n    tokens = nltk.word_tokenize(text)\n    for w in tokens:\n        finalTokens.append(stemmer.stem(w)) # change this to lemmatizer.lemmatize(w) for Lemmatizing\n    text = \" \".join(finalTokens)\n    return text\n\ntext_stemming = pd.DataFrame(columns=['TextBefore', 'TextAfter', 'Changed'])\ntext_stemming['TextBefore'] = text.copy()","metadata":{"_uuid":"432ef1f008675e538b49d9c4aec479647638c093","execution":{"iopub.status.busy":"2021-07-31T05:47:46.523904Z","iopub.execute_input":"2021-07-31T05:47:46.524247Z","iopub.status.idle":"2021-07-31T05:47:46.805680Z","shell.execute_reply.started":"2021-07-31T05:47:46.524218Z","shell.execute_reply":"2021-07-31T05:47:46.804681Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for index, row in text_stemming.iterrows():\n    row['TextAfter'] = tokenize(row['TextBefore'])","metadata":{"_uuid":"8c9b54b331016c0c7f41df1fd21181fb66a09121","execution":{"iopub.status.busy":"2021-07-31T05:47:46.807236Z","iopub.execute_input":"2021-07-31T05:47:46.807672Z","iopub.status.idle":"2021-07-31T06:02:51.309543Z","shell.execute_reply.started":"2021-07-31T05:47:46.807627Z","shell.execute_reply":"2021-07-31T06:02:51.307587Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"text_stemming['Changed'] = np.where(text_stemming['TextBefore'].str.replace(\" \",\"\")==text_stemming['TextAfter'].str.replace(\" \",\"\").str.replace(\"``\",'\"').str.replace(\"''\",'\"'), 'no', 'yes')\nprint(\"{} of {} ({:.4f}%) questions have been changed.\".format(len(text_stemming[text_stemming['Changed']=='yes']), len(text_stemming), 100*len(text_stemming[text_stemming['Changed']=='yes'])/len(text_stemming)))","metadata":{"_uuid":"8fa2ee8a86cc0a92088afca8a679ff0e63d461cb","execution":{"iopub.status.busy":"2021-07-31T06:02:51.312101Z","iopub.execute_input":"2021-07-31T06:02:51.312553Z","iopub.status.idle":"2021-07-31T06:02:59.663088Z","shell.execute_reply.started":"2021-07-31T06:02:51.312508Z","shell.execute_reply":"2021-07-31T06:02:59.661890Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for index, row in text_stemming[text_stemming['Changed']=='yes'].head().iterrows():\n    print(row['TextBefore'],'->',row['TextAfter'])","metadata":{"_uuid":"f602787bee9ab08e45590878f4f208c6d9092a25","execution":{"iopub.status.busy":"2021-07-31T06:02:59.664594Z","iopub.execute_input":"2021-07-31T06:02:59.664932Z","iopub.status.idle":"2021-07-31T06:02:59.967587Z","shell.execute_reply.started":"2021-07-31T06:02:59.664900Z","shell.execute_reply":"2021-07-31T06:02:59.966291Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Combos\nOf course we can use more than one technique at the same time. The order is essential here.\n\n**Example:** What are the recommended 2D game engines for a beginning Python programmer? -> what recommend d game engin begin python programm","metadata":{"_uuid":"38c7bd7939a3295d50551d229a4aa8ac7d48f42b"}},{"cell_type":"code","source":"def tokenize(text):\n    finalTokens = []\n    tokens = nltk.word_tokenize(text)\n    for w in tokens:\n        if (w not in stoplist):\n            w = addCapTag(w) # Handle Capitalized Words\n            w = w.lower() # Lowercase\n            w = replaceElongated(w) # Replace Elongated Words\n            w = stemmer.stem(w) # Stemming\n            finalTokens.append(w)\n    text = \" \".join(finalTokens)\n    return text\n\ntext_combos = pd.DataFrame(columns=['TextBefore', 'TextAfter', 'Changed'])\ntext_combos['TextBefore'] = text.copy()","metadata":{"_uuid":"9fa801a5bfdb1a57e96ac606474f50c74f840e70","execution":{"iopub.status.busy":"2021-07-31T06:02:59.969265Z","iopub.execute_input":"2021-07-31T06:02:59.969715Z","iopub.status.idle":"2021-07-31T06:03:00.539615Z","shell.execute_reply.started":"2021-07-31T06:02:59.969674Z","shell.execute_reply":"2021-07-31T06:03:00.538542Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for index, row in text_combos.iterrows():\n    row['TextAfter'] = replaceContraction(row['TextBefore']) # Replace Contractions\n    row['TextAfter'] = removeNumbers(row['TextAfter']) # Remove Integers\n    row['TextAfter'] = replaceMultiExclamationMark(row['TextAfter']) # Replace Multi Exclamation Marks\n    row['TextAfter'] = replaceMultiQuestionMark(row['TextAfter']) # Replace Multi Question Marks\n    row['TextAfter'] = replaceMultiStopMark(row['TextAfter']) # Repalce Multi Stop Marks\n    row['TextAfter'] = row['TextAfter'].translate(translator) # Remove Punctuation\n    row['TextAfter'] = tokenize(row['TextAfter'])","metadata":{"_uuid":"aac4438da5b434c25db3c0c5c7208b63ae4a0ad8","execution":{"iopub.status.busy":"2021-07-31T06:03:00.541267Z","iopub.execute_input":"2021-07-31T06:03:00.541680Z","iopub.status.idle":"2021-07-31T06:27:46.861002Z","shell.execute_reply.started":"2021-07-31T06:03:00.541639Z","shell.execute_reply":"2021-07-31T06:27:46.858370Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"text_combos['Changed'] = np.where(text_combos['TextBefore'].str.replace(\" \",\"\")==text_combos['TextAfter'].str.replace(\" \",\"\").str.replace(\"``\",'\"').str.replace(\"''\",'\"'), 'no', 'yes')\nprint(\"{} of {} ({:.4f}%) questions have been changed.\".format(len(text_combos[text_combos['Changed']=='yes']), len(text_combos), 100*len(text_combos[text_combos['Changed']=='yes'])/len(text_combos)))","metadata":{"_uuid":"a4d5cb507785c24065408f8dd17202a7accb24ca","execution":{"iopub.status.busy":"2021-07-31T06:27:46.864689Z","iopub.execute_input":"2021-07-31T06:27:46.865179Z","iopub.status.idle":"2021-07-31T06:27:54.582979Z","shell.execute_reply.started":"2021-07-31T06:27:46.865118Z","shell.execute_reply":"2021-07-31T06:27:54.581678Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for index, row in text_combos[text_combos['Changed']=='yes'].head().iterrows():\n    print(row['TextBefore'],'->',row['TextAfter'])","metadata":{"_uuid":"6607da645a7152b09a2dafe377351d8185326aa5","execution":{"iopub.status.busy":"2021-07-31T06:27:54.584429Z","iopub.execute_input":"2021-07-31T06:27:54.584858Z","iopub.status.idle":"2021-07-31T06:27:54.875753Z","shell.execute_reply.started":"2021-07-31T06:27:54.584822Z","shell.execute_reply":"2021-07-31T06:27:54.874526Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Thank you ","metadata":{"_uuid":"7df41aecf5dd758bfd0b80f08fa0a2533df66c29"}},{"cell_type":"code","source":"runs, running, ran ===> run(lemma)","metadata":{},"execution_count":null,"outputs":[]}]}