{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"raw","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19"}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport operator \nimport re\nfrom tqdm import tqdm","metadata":{"execution":{"iopub.status.busy":"2022-10-26T03:11:23.627437Z","iopub.execute_input":"2022-10-26T03:11:23.628306Z","iopub.status.idle":"2022-10-26T03:11:23.662023Z","shell.execute_reply.started":"2022-10-26T03:11:23.628176Z","shell.execute_reply":"2022-10-26T03:11:23.661126Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data = pd.read_csv('../input/quora-insincere-questions-classification/train.csv')\ntest_data = pd.read_csv('../input/quora-insincere-questions-classification/test.csv')","metadata":{"execution":{"iopub.status.busy":"2022-10-26T03:11:23.678373Z","iopub.execute_input":"2022-10-26T03:11:23.678644Z","iopub.status.idle":"2022-10-26T03:11:28.670097Z","shell.execute_reply.started":"2022-10-26T03:11:23.678619Z","shell.execute_reply":"2022-10-26T03:11:28.669039Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# test_data = pd.read_csv('../input/quora-insincere-questions-classification/test.csv')","metadata":{"execution":{"iopub.status.busy":"2022-10-26T03:11:28.672411Z","iopub.execute_input":"2022-10-26T03:11:28.672813Z","iopub.status.idle":"2022-10-26T03:11:28.677316Z","shell.execute_reply.started":"2022-10-26T03:11:28.672774Z","shell.execute_reply":"2022-10-26T03:11:28.676265Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.head()","metadata":{"execution":{"iopub.status.busy":"2022-10-26T03:11:28.679100Z","iopub.execute_input":"2022-10-26T03:11:28.679815Z","iopub.status.idle":"2022-10-26T03:11:28.704788Z","shell.execute_reply.started":"2022-10-26T03:11:28.679778Z","shell.execute_reply":"2022-10-26T03:11:28.703928Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"train data shape: \", train_data.shape)\nprint(\"test data shape: \", test_data.shape)","metadata":{"execution":{"iopub.status.busy":"2022-10-26T03:11:28.708644Z","iopub.execute_input":"2022-10-26T03:11:28.709499Z","iopub.status.idle":"2022-10-26T03:11:28.716913Z","shell.execute_reply.started":"2022-10-26T03:11:28.709455Z","shell.execute_reply":"2022-10-26T03:11:28.715864Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install contractions","metadata":{"execution":{"iopub.status.busy":"2022-10-26T03:11:28.718666Z","iopub.execute_input":"2022-10-26T03:11:28.719074Z","iopub.status.idle":"2022-10-26T03:11:39.703424Z","shell.execute_reply.started":"2022-10-26T03:11:28.719038Z","shell.execute_reply":"2022-10-26T03:11:39.702163Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import contractions\ndef expand_contractions(text):\n    return [contractions.fix(word) for word in text.split()]","metadata":{"execution":{"iopub.status.busy":"2022-10-26T03:11:39.705405Z","iopub.execute_input":"2022-10-26T03:11:39.706095Z","iopub.status.idle":"2022-10-26T03:11:39.731189Z","shell.execute_reply.started":"2022-10-26T03:11:39.706047Z","shell.execute_reply":"2022-10-26T03:11:39.730305Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data['question_text'] = train_data['question_text'].apply(lambda x: expand_contractions(x))\ntrain_data.head()","metadata":{"execution":{"iopub.status.busy":"2022-10-26T03:11:39.734283Z","iopub.execute_input":"2022-10-26T03:11:39.734559Z","iopub.status.idle":"2022-10-26T03:12:35.796547Z","shell.execute_reply.started":"2022-10-26T03:11:39.734534Z","shell.execute_reply":"2022-10-26T03:12:35.795645Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def join_words(column):\n    return (' '.join(column))","metadata":{"execution":{"iopub.status.busy":"2022-10-26T03:12:35.800493Z","iopub.execute_input":"2022-10-26T03:12:35.802670Z","iopub.status.idle":"2022-10-26T03:12:35.809162Z","shell.execute_reply.started":"2022-10-26T03:12:35.802632Z","shell.execute_reply":"2022-10-26T03:12:35.808098Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data['question_text'] = train_data['question_text'].apply(lambda x: join_words(x))\ntrain_data.head()","metadata":{"execution":{"iopub.status.busy":"2022-10-26T03:12:35.813018Z","iopub.execute_input":"2022-10-26T03:12:35.813667Z","iopub.status.idle":"2022-10-26T03:12:37.235156Z","shell.execute_reply.started":"2022-10-26T03:12:35.813631Z","shell.execute_reply":"2022-10-26T03:12:37.234162Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train_data['question_text'] = train_data['question_text'].apply(lambda x: x.lower())","metadata":{"execution":{"iopub.status.busy":"2022-10-26T03:12:37.239806Z","iopub.execute_input":"2022-10-26T03:12:37.240111Z","iopub.status.idle":"2022-10-26T03:12:37.245067Z","shell.execute_reply.started":"2022-10-26T03:12:37.240083Z","shell.execute_reply":"2022-10-26T03:12:37.243648Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def vocab_builder(texts, verbose=True):\n    vocab = {}\n    for text in tqdm(texts, disable = (not verbose)):\n        for word in text:\n            try:\n                vocab[word] += 1\n            except KeyError:\n                vocab[word] = 1\n    return vocab","metadata":{"execution":{"iopub.status.busy":"2022-10-26T03:12:37.246762Z","iopub.execute_input":"2022-10-26T03:12:37.247129Z","iopub.status.idle":"2022-10-26T03:12:37.256774Z","shell.execute_reply.started":"2022-10-26T03:12:37.247094Z","shell.execute_reply":"2022-10-26T03:12:37.255786Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sentences = train_data[\"question_text\"].apply(lambda x: x.split()).values\nvocab = vocab_builder(sentences)\nprint({k: vocab[k] for k in list(vocab)[:5]})","metadata":{"execution":{"iopub.status.busy":"2022-10-26T03:12:37.258227Z","iopub.execute_input":"2022-10-26T03:12:37.258622Z","iopub.status.idle":"2022-10-26T03:12:45.813418Z","shell.execute_reply.started":"2022-10-26T03:12:37.258587Z","shell.execute_reply":"2022-10-26T03:12:45.812270Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from zipfile import ZipFile\nwith ZipFile('../input/quora-insincere-questions-classification/embeddings.zip','r') as zipObj:\n    listOfFiles = zipObj.namelist()\nlistOfFiles","metadata":{"execution":{"iopub.status.busy":"2022-10-26T03:12:45.814943Z","iopub.execute_input":"2022-10-26T03:12:45.815434Z","iopub.status.idle":"2022-10-26T03:12:45.828636Z","shell.execute_reply.started":"2022-10-26T03:12:45.815395Z","shell.execute_reply":"2022-10-26T03:12:45.827046Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import zipfile\nfrom gensim.models import KeyedVectors\ndef load_embed(file):\n    def get_coefs(word,*arr): \n        return word, np.asarray(arr, dtype='float32')\n    \n    if file == '../input/embeddings/wiki-news-300d-1M/wiki-news-300d-1M.vec':\n        embeddings_index = dict(get_coefs(*o.split(\" \")) for o in open(file) if len(o)>100)\n    else:\n        embeddings_index = dict(get_coefs(*o.split(\" \")) for o in open(file, encoding='latin'))\n        \n    return embeddings_index","metadata":{"execution":{"iopub.status.busy":"2022-10-26T03:12:45.830251Z","iopub.execute_input":"2022-10-26T03:12:45.830610Z","iopub.status.idle":"2022-10-26T03:12:46.692126Z","shell.execute_reply.started":"2022-10-26T03:12:45.830575Z","shell.execute_reply":"2022-10-26T03:12:46.691145Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"glove = '../input/glove-embed/glove.840B.300d.txt'\nembed_glove = load_embed(glove)","metadata":{"execution":{"iopub.status.busy":"2022-10-26T03:12:46.693734Z","iopub.execute_input":"2022-10-26T03:12:46.694108Z","iopub.status.idle":"2022-10-26T03:15:02.413358Z","shell.execute_reply.started":"2022-10-26T03:12:46.694066Z","shell.execute_reply":"2022-10-26T03:15:02.412272Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import operator\n\ndef check_coverage(vocab,embeddings_index):\n    a = {}\n    oov = {}\n    k = 0\n    i = 0\n    for word in tqdm(vocab):\n        try:\n            a[word] = embeddings_index[word]\n            k += vocab[word]\n        except:\n\n            oov[word] = vocab[word]\n            i += vocab[word]\n            pass\n\n    print('Found embeddings for {:.2%} of vocab'.format(len(a) / len(vocab)))\n    print('Found embeddings for  {:.2%} of all text'.format(k / (k + i)))\n    sorted_x = sorted(oov.items(), key=operator.itemgetter(1))[::-1]\n\n    return sorted_x","metadata":{"execution":{"iopub.status.busy":"2022-10-26T03:15:02.415103Z","iopub.execute_input":"2022-10-26T03:15:02.415794Z","iopub.status.idle":"2022-10-26T03:15:02.424568Z","shell.execute_reply.started":"2022-10-26T03:15:02.415752Z","shell.execute_reply":"2022-10-26T03:15:02.423562Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"oov = check_coverage(vocab, embed_glove)","metadata":{"execution":{"iopub.status.busy":"2022-10-26T03:15:02.425959Z","iopub.execute_input":"2022-10-26T03:15:02.426430Z","iopub.status.idle":"2022-10-26T03:15:03.190043Z","shell.execute_reply.started":"2022-10-26T03:15:02.426393Z","shell.execute_reply":"2022-10-26T03:15:03.188938Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"oov[:10]","metadata":{"execution":{"iopub.status.busy":"2022-10-26T03:15:03.191798Z","iopub.execute_input":"2022-10-26T03:15:03.192481Z","iopub.status.idle":"2022-10-26T03:15:03.207459Z","shell.execute_reply.started":"2022-10-26T03:15:03.192438Z","shell.execute_reply":"2022-10-26T03:15:03.206136Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# def remove_punctuation(x):\n#     x = str(x)\n#     for punct in \"/-'\":\n#         x = x.replace(punct, ' ')\n#     for punct in '&':\n#         x = x.replace(punct, f' {punct} ')\n#     for punct in '?!.,\"#$%\\'()*+-/:;<=>@[\\\\]^_`{|}~' + '“”’':\n#         x = x.replace(punct, '')\n#     return x","metadata":{"execution":{"iopub.status.busy":"2022-10-26T03:15:03.209184Z","iopub.execute_input":"2022-10-26T03:15:03.209583Z","iopub.status.idle":"2022-10-26T03:15:03.217298Z","shell.execute_reply.started":"2022-10-26T03:15:03.209547Z","shell.execute_reply":"2022-10-26T03:15:03.216105Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train_data['question_text'] = train_data['question_text'].apply(lambda x: remove_punctuation(x))\n# sentences = train_data['question_text'].apply(lambda x: x.split())\n# vocab = vocab_builder(sentences)","metadata":{"execution":{"iopub.status.busy":"2022-10-26T03:15:03.218734Z","iopub.execute_input":"2022-10-26T03:15:03.219330Z","iopub.status.idle":"2022-10-26T03:15:03.229049Z","shell.execute_reply.started":"2022-10-26T03:15:03.219287Z","shell.execute_reply":"2022-10-26T03:15:03.228030Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# oov = check_coverage(vocab, embed_glove)","metadata":{"execution":{"iopub.status.busy":"2022-10-26T03:15:03.230572Z","iopub.execute_input":"2022-10-26T03:15:03.231042Z","iopub.status.idle":"2022-10-26T03:15:03.239514Z","shell.execute_reply.started":"2022-10-26T03:15:03.231002Z","shell.execute_reply":"2022-10-26T03:15:03.238405Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# oov[:10]","metadata":{"execution":{"iopub.status.busy":"2022-10-26T03:15:03.243116Z","iopub.execute_input":"2022-10-26T03:15:03.243431Z","iopub.status.idle":"2022-10-26T03:15:03.249264Z","shell.execute_reply.started":"2022-10-26T03:15:03.243405Z","shell.execute_reply":"2022-10-26T03:15:03.248212Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import re\n\n# def clean_numbers(x):\n\n#     x = re.sub('[0-9]{5,}', '#####', x)\n#     x = re.sub('[0-9]{4}', '####', x)\n#     x = re.sub('[0-9]{3}', '###', x)\n#     x = re.sub('[0-9]{2}', '##', x)\n#     return x","metadata":{"execution":{"iopub.status.busy":"2022-10-26T03:15:03.250543Z","iopub.execute_input":"2022-10-26T03:15:03.250949Z","iopub.status.idle":"2022-10-26T03:15:03.258908Z","shell.execute_reply.started":"2022-10-26T03:15:03.250912Z","shell.execute_reply":"2022-10-26T03:15:03.257849Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train_data[\"question_text\"] = train_data[\"question_text\"].apply(lambda x: clean_numbers(x))\n# sentences = train_data[\"question_text\"].apply(lambda x: x.split())\n# vocab = vocab_builder(sentences)","metadata":{"execution":{"iopub.status.busy":"2022-10-26T03:15:03.260077Z","iopub.execute_input":"2022-10-26T03:15:03.261150Z","iopub.status.idle":"2022-10-26T03:15:03.268738Z","shell.execute_reply.started":"2022-10-26T03:15:03.261112Z","shell.execute_reply":"2022-10-26T03:15:03.267689Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# oov = check_coverage(vocab, embed_glove)","metadata":{"execution":{"iopub.status.busy":"2022-10-26T03:15:03.271722Z","iopub.execute_input":"2022-10-26T03:15:03.272096Z","iopub.status.idle":"2022-10-26T03:15:03.282569Z","shell.execute_reply.started":"2022-10-26T03:15:03.272070Z","shell.execute_reply":"2022-10-26T03:15:03.281477Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# oov[:10]","metadata":{"execution":{"iopub.status.busy":"2022-10-26T03:15:03.283868Z","iopub.execute_input":"2022-10-26T03:15:03.284560Z","iopub.status.idle":"2022-10-26T03:15:03.292612Z","shell.execute_reply.started":"2022-10-26T03:15:03.284522Z","shell.execute_reply":"2022-10-26T03:15:03.291595Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# sentences = train_data[\"question_text\"].apply(lambda x: x.split())\n# to_remove = ['a','to','of','and']\n# sentences = [[word for word in sentence if not word in to_remove] for sentence in tqdm(sentences)]\n# vocab = vocab_builder(sentences)","metadata":{"execution":{"iopub.status.busy":"2022-10-26T03:15:03.294698Z","iopub.execute_input":"2022-10-26T03:15:03.294992Z","iopub.status.idle":"2022-10-26T03:15:03.303896Z","shell.execute_reply.started":"2022-10-26T03:15:03.294964Z","shell.execute_reply":"2022-10-26T03:15:03.302556Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# oov = check_coverage(vocab,embeddings_index)","metadata":{"execution":{"iopub.status.busy":"2022-10-26T03:15:03.314546Z","iopub.execute_input":"2022-10-26T03:15:03.314802Z","iopub.status.idle":"2022-10-26T03:15:03.319387Z","shell.execute_reply.started":"2022-10-26T03:15:03.314778Z","shell.execute_reply":"2022-10-26T03:15:03.318379Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def clean_number(text):\n    text = re.sub(r'(\\d+)([a-zA-Z])', '\\g<1> \\g<2>', text)\n    text = re.sub(r'(\\d+) (th|st|nd|rd) ', '\\g<1>\\g<2> ', text)\n    text = re.sub(r'(\\d+),(\\d+)', '\\g<1>\\g<2>', text)\n    \n    return text","metadata":{"execution":{"iopub.status.busy":"2022-10-26T03:15:03.320799Z","iopub.execute_input":"2022-10-26T03:15:03.321287Z","iopub.status.idle":"2022-10-26T03:15:03.328518Z","shell.execute_reply.started":"2022-10-26T03:15:03.321249Z","shell.execute_reply":"2022-10-26T03:15:03.327362Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import string\nregular_punct = list(string.punctuation)\nextra_punct = [\n    ',', '.', '\"', ':', ')', '(', '!', '?', '|', ';', \"'\", '$', '&',\n    '/', '[', ']', '>', '%', '=', '#', '*', '+', '\\\\', '•',  '~', '@', '£',\n    '·', '_', '{', '}', '©', '^', '®', '`',  '<', '→', '°', '€', '™', '›',\n    '♥', '←', '×', '§', '″', '′', 'Â', '█', '½', 'à', '…', '“', '★', '”',\n    '–', '●', 'â', '►', '−', '¢', '²', '¬', '░', '¶', '↑', '±', '¿', '▾',\n    '═', '¦', '║', '―', '¥', '▓', '—', '‹', '─', '▒', '：', '¼', '⊕', '▼',\n    '▪', '†', '■', '’', '▀', '¨', '▄', '♫', '☆', 'é', '¯', '♦', '¤', '▲',\n    'è', '¸', '¾', 'Ã', '⋅', '‘', '∞', '∙', '）', '↓', '、', '│', '（', '»',\n    '，', '♪', '╩', '╚', '³', '・', '╦', '╣', '╔', '╗', '▬', '❤', 'ï', 'Ø',\n    '¹', '≤', '‡', '√', '«', '»', '´', 'º', '¾', '¡', '§', '£', '₤']\nall_punct = list(set(regular_punct + extra_punct))\n# do not spacing - and .\nall_punct.remove('-')\nall_punct.remove('.')\n\ndef spacing_punctuation(text):\n    \"\"\"\n    add space before and after punctuation and symbols\n    \"\"\"\n    for punc in all_punct:\n        if punc in text:\n            text = text.replace(punc, f' {punc} ')\n    return text","metadata":{"execution":{"iopub.status.busy":"2022-10-26T03:15:03.330277Z","iopub.execute_input":"2022-10-26T03:15:03.330774Z","iopub.status.idle":"2022-10-26T03:15:03.343743Z","shell.execute_reply.started":"2022-10-26T03:15:03.330739Z","shell.execute_reply":"2022-10-26T03:15:03.342631Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# oov[:50]","metadata":{"execution":{"iopub.status.busy":"2022-10-26T03:15:03.345650Z","iopub.execute_input":"2022-10-26T03:15:03.346083Z","iopub.status.idle":"2022-10-26T03:15:03.357156Z","shell.execute_reply.started":"2022-10-26T03:15:03.346047Z","shell.execute_reply":"2022-10-26T03:15:03.356185Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# misspelt_dict = {'favourite': 'favorite', 'colour':'color', 'centre':'center', 'bitcoin':'digital currency',\n#                 'instagram':'social media', 'upsc':'union public service commission', 'btech': 'bachelor of technology',\n#                 'mbbs':'bachelor of medicine bachelor of surgery', 'whatsapp':'social media', 'ece':'electronics and communications engineering',\n#                 'aiims':'all india institute of medical science', 'iim':'indian institute of management', 'mtech':'masters in technology',\n#                 'sbi':'state bank of india','cgl':'combined graduate level', 'cryptocurrency':'digital currency', 'snapchat':'social media',\n#                 'obc':'other backward classes', 'jio':'network provider', 'manipal':'city', 'travelling':'traveling',\n#                 'bba':'bachelor of business administration', 'icse':'indian certificate of secondary education','counselling':'counseling',\n#                 'tcs':'tata consultancy services', 'srm':'college', 'wwii':'world war', 'cgpa':'cumulative grade point average',\n#                 'bitsat':'exam', 'iiit':'indian institute of information technology', 'iiit':'english proficiency test', 'brexit':'britain exit',\n#                 'iitians':'indian institute of technology', 'iit':'indian institute of technology', 'cryptocurrencies':'digital currency',\n#                 'ncert':'national council of educational research and training', 'behaviour':'behavior','programme':'program',\n#                 'clat':'common law admission test', 'isro':'indian space research organisation', 'bcom':'bachelor of commerce',\n#                 'upvotes':'positive vote', 'bca':'bachelor in computer application', 'defence':'defense', 'grey':'gray',\n#                 'isc':'indian school certificate','theatre': 'theater', 'cancelled': 'canceled', 'labour': 'labor', 'organisation': 'organization',\n#                 'citicise': 'criticize', 'youtu ': 'youtube ', 'Qoura': 'Quora', 'sallary': 'salary', 'Whta': 'What', 'narcisist': 'narcissist', \n#                  'howdo': 'how do', 'whatare': 'what are', 'howcan': 'how can', 'howmuch': 'how much', 'howmany': 'how many',\n#                 'whydo': 'why do', 'doI': 'do I', 'theBest': 'the best', 'howdoes': 'how does', 'mastrubation': 'masturbation',\n#                 'mastrubate': 'masturbate', \"mastrubating\": 'masturbating', 'pennis': 'penis', 'Etherium': 'Ethereum',\n#                 'narcissit': 'narcissist', 'bigdata': 'big data', '2k17': '2017', '2k18': '2018', 'qouta': 'quota',\n#                 'exboyfriend': 'ex boyfriend', 'airhostess': 'air hostess', \"whst\": 'what', 'watsapp': 'whatsapp', \n#                 'demonitisation': 'demonetization', 'demonitization': 'demonetization', 'demonetisation': 'demonetization',\n#                 'relectronics':'electronics','dindian':'indian','nelectronics':'electronics','engineeringssary':'engineering',\n#                 'engineeringnt':'engineering', 'delectronics':'electronics','engineeringntly':'engineering','pokémon':'pokemon','redmi':'phone brand'}","metadata":{"execution":{"iopub.status.busy":"2022-10-26T03:15:03.360468Z","iopub.execute_input":"2022-10-26T03:15:03.360747Z","iopub.status.idle":"2022-10-26T03:15:03.367506Z","shell.execute_reply.started":"2022-10-26T03:15:03.360722Z","shell.execute_reply":"2022-10-26T03:15:03.366260Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# def correct_spelling(x, dic):\n#     for word in dic.keys():\n#         x = x.replace(word, dic[word])\n#     return x","metadata":{"execution":{"iopub.status.busy":"2022-10-26T03:15:03.368816Z","iopub.execute_input":"2022-10-26T03:15:03.369327Z","iopub.status.idle":"2022-10-26T03:15:03.381494Z","shell.execute_reply.started":"2022-10-26T03:15:03.369291Z","shell.execute_reply":"2022-10-26T03:15:03.380518Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train_data['question_text'] = train_data['question_text'].apply(lambda x: correct_spelling(x, misspelt_dict))\n# train_data.head()","metadata":{"execution":{"iopub.status.busy":"2022-10-26T03:15:03.382754Z","iopub.execute_input":"2022-10-26T03:15:03.383220Z","iopub.status.idle":"2022-10-26T03:15:03.391257Z","shell.execute_reply.started":"2022-10-26T03:15:03.383162Z","shell.execute_reply":"2022-10-26T03:15:03.390097Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# sentences = train_data[\"question_text\"].apply(lambda x: x.split()).values\n# vocab = vocab_builder(sentences)","metadata":{"execution":{"iopub.status.busy":"2022-10-26T03:15:03.394278Z","iopub.execute_input":"2022-10-26T03:15:03.395157Z","iopub.status.idle":"2022-10-26T03:15:03.401263Z","shell.execute_reply.started":"2022-10-26T03:15:03.395117Z","shell.execute_reply":"2022-10-26T03:15:03.400252Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# oov = check_coverage(vocab,embed_glove)","metadata":{"execution":{"iopub.status.busy":"2022-10-26T03:15:03.403014Z","iopub.execute_input":"2022-10-26T03:15:03.403481Z","iopub.status.idle":"2022-10-26T03:15:03.413530Z","shell.execute_reply.started":"2022-10-26T03:15:03.403444Z","shell.execute_reply":"2022-10-26T03:15:03.412610Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"oov[:10]","metadata":{"execution":{"iopub.status.busy":"2022-10-26T03:15:03.415483Z","iopub.execute_input":"2022-10-26T03:15:03.415888Z","iopub.status.idle":"2022-10-26T03:15:03.426170Z","shell.execute_reply.started":"2022-10-26T03:15:03.415853Z","shell.execute_reply":"2022-10-26T03:15:03.425188Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# sentences = train_data[\"question_text\"].apply(lambda x: x.split())\n# to_remove = ['a','to','of','and']\n# sentences = [[word for word in sentence if not word in to_remove] for sentence in tqdm(sentences)]\n# vocab = vocab_builder(sentences)","metadata":{"execution":{"iopub.status.busy":"2022-10-26T03:15:03.427527Z","iopub.execute_input":"2022-10-26T03:15:03.427985Z","iopub.status.idle":"2022-10-26T03:15:03.435075Z","shell.execute_reply.started":"2022-10-26T03:15:03.427951Z","shell.execute_reply":"2022-10-26T03:15:03.433991Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# oov = check_coverage(vocab,embeddings_index)","metadata":{"execution":{"iopub.status.busy":"2022-10-26T03:15:03.436901Z","iopub.execute_input":"2022-10-26T03:15:03.437269Z","iopub.status.idle":"2022-10-26T03:15:03.444573Z","shell.execute_reply.started":"2022-10-26T03:15:03.437234Z","shell.execute_reply":"2022-10-26T03:15:03.443359Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# oov[:50]","metadata":{"execution":{"iopub.status.busy":"2022-10-26T03:15:03.448177Z","iopub.execute_input":"2022-10-26T03:15:03.448530Z","iopub.status.idle":"2022-10-26T03:15:03.454236Z","shell.execute_reply.started":"2022-10-26T03:15:03.448502Z","shell.execute_reply":"2022-10-26T03:15:03.453286Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"spaces = ['\\u200b', '\\u200e', '\\u202a', '\\u202c', '\\ufeff', '\\uf0d8', '\\u2061', '\\x10', '\\x7f', '\\x9d', '\\xad', '\\xa0']\ndef remove_space(text):\n    \"\"\"\n    remove extra spaces and ending space if any\n    \"\"\"\n    for space in spaces:\n        text = text.replace(space, ' ')\n    text = text.strip()\n    text = re.sub('\\s+', ' ', text)\n    return text","metadata":{"execution":{"iopub.status.busy":"2022-10-26T03:15:03.456224Z","iopub.execute_input":"2022-10-26T03:15:03.456664Z","iopub.status.idle":"2022-10-26T03:15:03.464170Z","shell.execute_reply.started":"2022-10-26T03:15:03.456627Z","shell.execute_reply":"2022-10-26T03:15:03.462782Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from unicodedata import category, name, normalize\n\ndef remove_diacritics(s):\n    return ''.join(c for c in normalize('NFKD', s.replace('ø', 'o').replace('Ø', 'O').replace('⁻', '-').replace('₋', '-'))\n                  if category(c) != 'Mn')\n\nspecial_punc_mappings = {\"—\": \"-\", \"–\": \"-\", \"_\": \"-\", '”': '\"', \"″\": '\"', '“': '\"', '•': '.', '−': '-',\n                         \"’\": \"'\", \"‘\": \"'\", \"´\": \"'\", \"`\": \"'\", '\\u200b': ' ', '\\xa0': ' ','،':'','„':'',\n                         '…': ' ... ', '\\ufeff': ''}\ndef clean_special_punctuations(text):\n    for punc in special_punc_mappings:\n        if punc in text:\n            text = text.replace(punc, special_punc_mappings[punc])\n    text = remove_diacritics(text)\n    return text","metadata":{"execution":{"iopub.status.busy":"2022-10-26T03:15:03.465583Z","iopub.execute_input":"2022-10-26T03:15:03.466156Z","iopub.status.idle":"2022-10-26T03:15:03.484046Z","shell.execute_reply.started":"2022-10-26T03:15:03.466118Z","shell.execute_reply":"2022-10-26T03:15:03.483006Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"bad_words_mapping = {' s.p ': ' ', ' S.P ': ' ', 'U.s.p': '', 'U.S.A.': 'USA', 'u.s.a.': 'USA', 'U.S.A': 'USA',\n                      'u.s.a': 'USA', 'U.S.': 'USA', 'u.s.': 'USA', ' U.S ': ' USA ', ' u.s ': ' USA ', 'U.s.': 'USA',\n                      ' U.s ': 'USA', ' u.S ': ' USA ', 'fu.k': 'fuck', 'U.K.': 'UK', ' u.k ': ' UK ',\n                      ' don t ': ' do not ', 'bacteries': 'batteries', ' yr old ': ' years old ', 'Ph.D': 'PhD',\n                      'cau.sing': 'causing', 'Kim Jong-Un': 'The president of North Korea', 'savegely': 'savagely',\n                      'Ra apist': 'Rapist', '2fifth': 'twenty fifth', '2third': 'twenty third',\n                      '2nineth': 'twenty nineth', '2fourth': 'twenty fourth', '#metoo': 'MeToo',\n                      'Trumpcare': 'Trump health care system', '4fifth': 'forty fifth', 'Remainers': 'remainder',\n                      'Terroristan': 'terrorist', 'antibrahmin': 'anti brahmin',\n                      'fuckboys': 'fuckboy', 'Fuckboys': 'fuckboy', 'Fuckboy': 'fuckboy', 'fuckgirls': 'fuck girls',\n                      'fuckgirl': 'fuck girl', 'Trumpsters': 'Trump supporters', '4sixth': 'forty sixth',\n                      'culturr': 'culture',\n                      'weatern': 'western', '4fourth': 'forty fourth', 'emiratis': 'emirates', 'trumpers': 'Trumpster',\n                      'indans': 'indians', 'mastuburate': 'masturbate', 'f**k': 'fuck', 'F**k': 'fuck', 'F**K': 'fuck',\n                      ' u r ': ' you are ', ' u ': ' you ', '操你妈': 'fuck your mother', 'e.g.': 'for example',\n                      'i.e.': 'in other words', '...': '.', 'et.al': 'elsewhere', 'anti-Semitic': 'anti-semitic',\n                      'f***': 'fuck', 'f**': 'fuc', 'F***': 'fuck', 'F**': 'fuc',\n                      'a****': 'assho', 'a**': 'ass', 'h***': 'hole', 'A****': 'assho', 'A**': 'ass', 'H***': 'hole',\n                      's***': 'shit', 's**': 'shi', 'S***': 'shit', 'S**': 'shi', 'Sh**': 'shit',\n                      'p****': 'pussy', 'p*ssy': 'pussy', 'P****': 'pussy',\n                      'p***': 'porn', 'p*rn': 'porn', 'P***': 'porn',\n                      'st*up*id': 'stupid',\n                      'd***': 'dick', 'di**': 'dick', 'h*ck': 'hack',\n                      'b*tch': 'bitch', 'bi*ch': 'bitch', 'bit*h': 'bitch', 'bitc*': 'bitch', 'b****': 'bitch',\n                      'b***': 'bitc', 'b**': 'bit', 'b*ll': 'bull',\"YOUSA\":'USA'\n                      }\n\n\ndef pre_clean_bad_words(text):\n    for bad_word in bad_words_mapping:\n        if bad_word in text:\n            text = text.replace(bad_word, bad_words_mapping[bad_word])\n\n    return text","metadata":{"execution":{"iopub.status.busy":"2022-10-26T03:15:03.485640Z","iopub.execute_input":"2022-10-26T03:15:03.486078Z","iopub.status.idle":"2022-10-26T03:15:03.499534Z","shell.execute_reply.started":"2022-10-26T03:15:03.486042Z","shell.execute_reply":"2022-10-26T03:15:03.498408Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def clean_latex(text):\n    \"\"\"\n    convert r\"[math]\\vec{x} + \\vec{y}\" to English\n    \"\"\"\n    # edge case\n    text = re.sub(r'\\[math\\]', ' LaTex math ', text)\n    text = re.sub(r'\\[\\/math\\]', ' LaTex math ', text)\n    text = re.sub(r'\\\\', ' LaTex ', text)\n\n    pattern_to_sub = {\n        r'\\\\mathrm': ' LaTex math mode ',\n        r'\\\\mathbb': ' LaTex math mode ',\n        r'\\\\boxed': ' LaTex equation ',\n        r'\\\\begin': ' LaTex equation ',\n        r'\\\\end': ' LaTex equation ',\n        r'\\\\left': ' LaTex equation ',\n        r'\\\\right': ' LaTex equation ',\n        r'\\\\(over|under)brace': ' LaTex equation ',\n        r'\\\\text': ' LaTex equation ',\n        r'\\\\vec': ' vector ',\n        r'\\\\var': ' variable ',\n        r'\\\\theta': ' theta ',\n        r'\\\\mu': ' average ',\n        r'\\\\min': ' minimum ',\n        r'\\\\max': ' maximum ',\n        r'\\\\sum': ' + ',\n        r'\\\\times': ' * ',\n        r'\\\\cdot': ' * ',\n        r'\\\\hat': ' ^ ',\n        r'\\\\frac': ' / ',\n        r'\\\\div': ' / ',\n        r'\\\\sin': ' Sine ',\n        r'\\\\cos': ' Cosine ',\n        r'\\\\tan': ' Tangent ',\n        r'\\\\infty': ' infinity ',\n        r'\\\\int': ' integer ',\n        r'\\\\in': ' in ',\n    }\n    # post process for look up\n    pattern_dict = {k.strip('\\\\'): v for k, v in pattern_to_sub.items()}\n    # init re\n    patterns = pattern_to_sub.keys()\n    pattern_re = re.compile('(%s)' % '|'.join(patterns))\n\n    def _replace(match):\n        try:\n            word = pattern_dict.get(match.group(0).strip('\\\\'))\n        except KeyError:\n            word = match.group(0)\n            print('!!Error: Could Not Find Key: {}'.format(word))\n        return word\n    return pattern_re.sub(_replace, text)","metadata":{"execution":{"iopub.status.busy":"2022-10-26T03:15:03.501266Z","iopub.execute_input":"2022-10-26T03:15:03.501627Z","iopub.status.idle":"2022-10-26T03:15:03.514463Z","shell.execute_reply.started":"2022-10-26T03:15:03.501586Z","shell.execute_reply":"2022-10-26T03:15:03.513808Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"misspell_mapping = {'Terroristan': 'terrorist Pakistan', 'terroristan': 'terrorist Pakistan',\n                    'FATF': 'Western summit conference',\n                    'BIMARU': 'BIMARU Bihar, Madhya Pradesh, Rajasthan, Uttar Pradesh', 'Hinduphobic': 'Hindu phobic',\n                    'hinduphobic': 'Hindu phobic', 'Hinduphobia': 'Hindu phobic', 'hinduphobia': 'Hindu phobic',\n                    'Babchenko': 'Arkady Arkadyevich Babchenko faked death', 'Boshniaks': 'Bosniaks',\n                    'Dravidanadu': 'Dravida Nadu', 'mysoginists': 'misogynists', 'MGTOWS': 'Men Going Their Own Way',\n                    'mongloid': 'Mongoloid', 'unsincere': 'insincere', 'meninism': 'male feminism',\n                    'jewplicate': 'jewish replicate', 'jewplicates': 'jewish replicate', 'andhbhakts': 'and Bhakt',\n                    'unoin': 'Union', 'daesh': 'Islamic State of Iraq and the Levant', 'burnol': 'movement about Modi',\n                    'Kalergi': 'Coudenhove-Kalergi', 'Bhakts': 'Bhakt', 'bhakts': 'Bhakt', 'Tambrahms': 'Tamil Brahmin',\n                    'Pahul': 'Amrit Sanskar', 'SJW': 'social justice warrior', 'SJWs': 'social justice warrior',\n                    ' incel': ' involuntary celibates', ' incels': ' involuntary celibates', 'emiratis': 'Emiratis',\n                    'weatern': 'western', 'westernise': 'westernize', 'Pizzagate': 'debunked conspiracy theory',\n                    'naïve': 'naive', 'Skripal': 'Russian military officer', 'Skripals': 'Russian military officer',\n                    'Remainers': 'British remainer', 'Novichok': 'Soviet Union agents',\n                    'gauri lankesh': 'Famous Indian Journalist', 'Castroists': 'Castro supporters',\n                    'remainers': 'British remainer', 'bremainer': 'British remainer', 'antibrahmin': 'anti Brahminism',\n                    'HYPSM': ' Harvard, Yale, Princeton, Stanford, MIT', 'HYPS': ' Harvard, Yale, Princeton, Stanford',\n                    'kompromat': 'compromising material', 'Tharki': 'pervert', 'tharki': 'pervert',\n                    'mastuburate': 'masturbate', 'Zoë': 'Zoe', 'indans': 'Indian', ' xender': ' gender',\n                    'Naxali ': 'Naxalite ', 'Naxalities': 'Naxalites', 'Bathla': 'Namit Bathla',\n                    'Mewani': 'Indian politician Jignesh Mevani', 'Wjy': 'Why',\n                    'Fadnavis': 'Indian politician Devendra Fadnavis', 'Awadesh': 'Indian engineer Awdhesh Singh',\n                    'Awdhesh': 'Indian engineer Awdhesh Singh', 'Khalistanis': 'Sikh separatist movement',\n                    'madheshi': 'Madheshi', 'BNBR': 'Be Nice, Be Respectful',\n                    'Jair Bolsonaro': 'Brazilian President politician', 'XXXTentacion': 'Tentacion',\n                    'Slavoj Zizek': 'Slovenian philosopher',\n                    'borderliners': 'borderlines', 'Brexit': 'British Exit', 'Brexiter': 'British Exit supporter',\n                    'Brexiters': 'British Exit supporters', 'Brexiteer': 'British Exit supporter',\n                    'Brexiteers': 'British Exit supporters', 'Brexiting': 'British Exit',\n                    'Brexitosis': 'British Exit disorder', 'brexit': 'British Exit',\n                    'brexiters': 'British Exit supporters', 'jallikattu': 'Jallikattu', 'fortnite': 'Fortnite',\n                    'Swachh': 'Swachh Bharat mission campaign ', 'Quorans': 'Quora users', 'Qoura': 'Quora',\n                    'quoras': 'Quora', 'Quroa': 'Quora', 'QUORA': 'Quora', 'Stupead': 'stupid',\n                    'narcissit': 'narcissist', 'trigger nometry': 'trigonometry',\n                    'trigglypuff': 'student Criticism of Conservatives', 'peoplelook': 'people look',\n                    'paedophelia': 'paedophilia', 'Uogi': 'Yogi', 'adityanath': 'Adityanath',\n                    'Yogi Adityanath': 'Indian monk and Hindu nationalist politician',\n                    'Awdhesh Singh': 'Commissioner of India', 'Doklam': 'Tibet', 'Drumpf ': 'Donald Trump fool ',\n                    'Drumpfs': 'Donald Trump fools', 'Strzok': 'Hillary Clinton scandal', 'rohingya': 'Rohingya ',\n                    ' wumao ': ' cheap Chinese stuff ', 'wumaos': 'cheap Chinese stuff', 'Sanghis': 'Sanghi',\n                    'Tamilans': 'Tamils', 'biharis': 'Biharis', 'Rejuvalex': 'hair growth formula Medicine',\n                    'Fekuchand': 'PM Narendra Modi in India', 'feku': 'Feku', 'Chaiwala': 'tea seller in India',\n                    'Feku': 'PM Narendra Modi in India ', 'deplorables': 'deplorable', 'muhajirs': 'Muslim immigrant',\n                    'Gujratis': 'Gujarati', 'Chutiya': 'Tibet people ', 'Chutiyas': 'Tibet people ',\n                    'thighing': 'masterbate between the legs of a female infant', '卐': 'Nazi Germany',\n                    'Pribumi': 'Native Indonesian', 'Gurmehar': 'Gurmehar Kaur Indian student activist',\n                    'Khazari': 'Khazars', 'Demonetization': 'demonetization', 'demonetisation': 'demonetization',\n                    'demonitisation': 'demonetization', 'demonitization': 'demonetization',\n                    'antinationals': 'antinational', 'Cryptocurrencies': 'cryptocurrency',\n                    'cryptocurrencies': 'cryptocurrency', 'Hindians': 'North Indian', 'Hindian': 'North Indian',\n                    'vaxxer': 'vocal nationalist ', 'remoaner': 'remainer ', 'bremoaner': 'British remainer ',\n                    'Jewism': 'Judaism', 'Eroupian': 'European', \"J & K Dy CM H ' ble Kavinderji\": '',\n                    'WMAF': 'White male married Asian female', 'AMWF': 'Asian male married White female',\n                    'moeslim': 'Muslim', 'cishet': 'cisgender and heterosexual person', 'Eurocentrics': 'Eurocentrism',\n                    'Eurocentric': 'Eurocentrism', 'Afrocentrics': 'Africa centrism', 'Afrocentric': 'Africa centrism',\n                    'Jewdar': 'Jew dar', 'marathis': 'Marathi', 'Gynophobic': 'Gyno phobic',\n                    'Trumpanzees': 'Trump chimpanzee fool', 'Crimean': 'Crimea people ', 'atrracted': 'attract',\n                    'Myeshia': 'widow of Green Beret killed in Niger', 'demcoratic': 'Democratic', 'raaping': 'raping',\n                    'feminazism': 'feminism nazi', 'langague': 'language', 'sathyaraj': 'actor',\n                    'Hongkongese': 'HongKong people', 'hongkongese': 'HongKong people', 'Kashmirians': 'Kashmirian',\n                    'Chodu': 'fucker', 'penish': 'penis',\n                    'chitpavan konkanastha': 'Hindu Maharashtrian Brahmin community',\n                    'Madridiots': 'Real Madrid idiot supporters', 'Ambedkarite': 'Dalit Buddhist movement ',\n                    'ReleaseTheMemo': 'cry for the right and Trump supporters', 'harrase': 'harass',\n                    'Barracoon': 'Black slave', 'Castrater': 'castration', 'castrater': 'castration',\n                    'Rapistan': 'Pakistan rapist', 'rapistan': 'Pakistan rapist', 'Turkified': 'Turkification',\n                    'turkified': 'Turkification', 'Dumbassistan': 'dumb ass Pakistan', 'facetards': 'Facebook retards',\n                    'rapefugees': 'rapist refugee', 'Khortha': 'language in the Indian state of Jharkhand',\n                    'Magahi': 'language in the northeastern Indian', 'Bajjika': 'language spoken in eastern India',\n                    'superficious': 'superficial', 'Sense8': 'American science fiction drama web television series',\n                    'Saipul Jamil': 'Indonesia artist', 'bhakht': 'bhakti', 'Smartia': 'dumb nation',\n                    'absorve': 'absolve', 'citicise': 'criticize', 'youtu ': 'youtube ', 'Whta': 'What',\n                    'esspecial': 'especial', 'doI': 'do I', 'theBest': 'the best',\n                    'howdoes': 'how does', 'Etherium': 'Ethereum', '2k17': '2017', '2k18': '2018', 'qiblas': 'qibla',\n                    'Hello4 2 cab': 'Online Cab Booking', 'bodyshame': 'body shaming', 'bodyshoppers': 'body shopping',\n                    'bodycams': 'body cams', 'Cananybody': 'Can any body', 'deadbody': 'dead body',\n                    'deaddict': 'de addict', 'Northindian': 'North Indian ', 'northindian': 'north Indian ',\n                    'northkorea': 'North Korea', 'koreaboo': 'Korea boo ',\n                    'Brexshit': 'British Exit bullshit', 'shitpost': 'shit post', 'shitslam': 'shit Islam',\n                    'shitlords': 'shit lords', 'Fck': 'Fuck', 'Clickbait': 'click bait ', 'clickbait': 'click bait ',\n                    'mailbait': 'mail bait', 'healhtcare': 'healthcare', 'trollbots': 'troll bots',\n                    'trollled': 'trolled', 'trollimg': 'trolling', 'cybertrolling': 'cyber trolling',\n                    'sickular': 'India sick secular ', 'Idiotism': 'idiotism',\n                    'Niggerism': 'Nigger', 'Niggeriah': 'Nigger'}\n\ndef clean_misspell(text):\n    for bad_word in misspell_mapping:\n        if bad_word in text:\n            text = text.replace(bad_word, misspell_mapping[bad_word])\n    return text","metadata":{"execution":{"iopub.status.busy":"2022-10-26T03:15:03.515971Z","iopub.execute_input":"2022-10-26T03:15:03.516619Z","iopub.status.idle":"2022-10-26T03:15:03.540244Z","shell.execute_reply.started":"2022-10-26T03:15:03.516585Z","shell.execute_reply":"2022-10-26T03:15:03.539242Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"bad_case_words = {'jewprofits': 'jew profits', 'QMAS': 'Quality Migrant Admission Scheme', 'casterating': 'castrating',\n                  'Kashmiristan': 'Kashmir', 'CareOnGo': 'India first and largest Online distributor of medicines',\n                  'Setya Novanto': 'a former Indonesian politician', 'TestoUltra': 'male sexual enhancement supplement',\n                  'rammayana': 'ramayana', 'Badaganadu': 'Brahmin community that mainly reside in Karnataka',\n                  'bitcjes': 'bitches', 'mastubrate': 'masturbate', 'Français': 'France',\n                  'Adsresses': 'address', 'flemmings': 'flemming', 'intermate': 'inter mating', 'feminisam': 'feminism',\n                  'cuckholdry': 'cuckold', 'Niggor': 'black hip-hop and electronic artist', 'narcsissist': 'narcissist',\n                  'Genderfluid': 'Gender fluid', ' Im ': ' I am ', ' dont ': ' do not ', 'Qoura': 'Quora',\n                  'ethethnicitesnicites': 'ethnicity', 'Namit Bathla': 'Content Writer', 'What sApp': 'WhatsApp',\n                  'Führer': 'Fuhrer', 'covfefe': 'coverage', 'accedentitly': 'accidentally', 'Cuckerberg': 'Zuckerberg',\n                  'transtrenders': 'incredibly disrespectful to real transgender people',\n                  'frozen tamod': 'Pornographic website', 'hindians': 'North Indian', 'hindian': 'North Indian',\n                  'celibatess': 'celibates', 'Trimp': 'Trump', 'wanket': 'wanker', 'wouldd': 'would',\n                  'arragent': 'arrogant', 'Ra - apist': 'rapist', 'idoot': 'idiot', 'gangstalkers': 'gangs talkers',\n                  'toastsexual': 'toast sexual', 'inapropriately': 'inappropriately', 'dumbassess': 'dumbass',\n                  'germanized': 'become german', 'helisexual': 'sexual', 'regilious': 'religious',\n                  'timetraveller': 'time traveller', 'darkwebcrawler': 'dark webcrawler', 'routez': 'route',\n                  'trumpians': 'Trump supporters', 'irreputable': 'reputation', 'serieusly': 'seriously',\n                  'anti cipation': 'anticipation', 'microaggression': 'micro aggression', 'Afircans': 'Africans',\n                  'microapologize': 'micro apologize', 'Vishnus': 'Vishnu', 'excritment': 'excitement',\n                  'disagreemen': 'disagreement', 'gujratis': 'gujarati', 'gujaratis': 'gujarati',\n                  'ugggggggllly': 'ugly',\n                  'Germanity': 'German', 'SoyBoys': 'cuck men lacking masculine characteristics',\n                  'н': 'h', 'м': 'm', 'ѕ': 's', 'т': 't', 'в': 'b', 'υ': 'u', 'ι': 'i',\n                  'genetilia': 'genitalia', 'r - apist': 'rapist', 'Borokabama': 'Barack Obama',\n                  'arectifier': 'rectifier', 'pettypotus': 'petty potus', 'magibabble': 'magi babble',\n                  'nothinking': 'thinking', 'centimiters': 'centimeters', 'saffronized': 'India, politics, derogatory',\n                  'saffronize': 'India, politics, derogatory', ' incect ': ' insect ', 'weenus': 'elbow skin',\n                  'Pakistainies': 'Pakistanis', 'goodspeaks': 'good speaks', 'inpregnated': 'in pregnant',\n                  'rapefilms': 'rape films', 'rapiest': 'rapist', 'hatrednesss': 'hatred',\n                  'heightism': 'height discrimination', 'getmy': 'get my', 'onsocial': 'on social',\n                  'worstplatform': 'worst platform', 'platfrom': 'platform', 'instagate': 'instigate',\n                  'Loy Machedeo': 'person', ' dsire ': ' desire ', 'iservant': 'servant', 'intelliegent': 'intelligent',\n                  'WW 1': ' WW1 ', 'WW 2': ' WW2 ', 'ww 1': ' WW1 ', 'ww 2': ' WW2 ',\n                  'keralapeoples': 'kerala peoples', 'trumpervotes': 'trumper votes', 'fucktrumpet': 'fuck trumpet',\n                  'likebJaish': 'like bJaish', 'likemy': 'like my', 'Howlikely': 'How likely',\n                  'disagreementts': 'disagreements', 'disagreementt': 'disagreement',\n                  'meninist': \"male chauvinism\", 'feminists': 'feminism supporters', 'Ghumendra': 'Bhupendra',\n                  'emellishments': 'embellishments',\n                  'settelemen': 'settlement',\n                  'Richmencupid': 'rich men dating website', 'richmencupid': 'rich men dating website',\n                  'Gaudry - Schost': '', 'ladymen': 'ladyboy', 'hasserment': 'Harassment',\n                  'instrumentalizing': 'instrument', 'darskin': 'dark skin', 'balckwemen': 'balck women',\n                  'recommendor': 'recommender', 'wowmen': 'women', 'expertthink': 'expert think',\n                  'whitesplaining': 'white splaining', 'Inquoraing': 'inquiring', 'whilemany': 'while many',\n                  'manyother': 'many other', 'involvedinthe': 'involved in the', 'slavetrade': 'slave trade',\n                  'aswell': 'as well', 'fewshowanyRemorse': 'few show any Remorse', 'trageting': 'targeting',\n                  'getile': 'gentile', 'Gujjus': 'derogatory Gujarati', 'judisciously': 'judiciously',\n                  'Hue Mungus': 'feminist bait', 'Hugh Mungus': 'feminist bait', 'Hindustanis': '',\n                  'Virushka': 'Great Relationships Couple', 'exclusinary': 'exclusionary', 'himdus': 'hindus',\n                  'Milo Yianopolous': 'a British polemicist', 'hidusim': 'hinduism',\n                  'holocaustable': 'holocaust', 'evangilitacal': 'evangelical', 'Busscas': 'Buscas',\n                  'holocaustal': 'holocaust', 'incestious': 'incestuous', 'Tennesseus': 'Tennessee',\n                  'GusDur': 'Gus Dur',\n                  'RPatah - Tan Eng Hwan': 'Silsilah', 'Reinfectus': 'reinfect', 'pharisaistic': 'pharisaism',\n                  'nuslims': 'Muslims', 'taskus': '', 'musims': 'Muslims',\n                  'Musevi': 'the independence of Mexico', ' racious ': 'discrimination expression of racism',\n                  'Muslimophobia': 'Muslim phobia', 'justyfied': 'justified', 'holocause': 'holocaust',\n                  'musilim': 'Muslim', 'misandrous': 'misandry', 'glrous': 'glorious', 'desemated': 'decimated',\n                  'votebanks': 'vote banks', 'Parkistan': 'Pakistan', 'Eurooe': 'Europe', 'animlaistic': 'animalistic',\n                  'Asiasoid': 'Asian', 'Congoid': 'Congolese', 'inheritantly': 'inherently',\n                  'Asianisation': 'Becoming Asia',\n                  'Russosphere': 'russia sphere of influence', 'exMuslims': 'Ex-Muslims',\n                  'discriminatein': 'discrimination', ' hinus ': ' hindus ', 'Nibirus': 'Nibiru',\n                  'habius - corpus': 'habeas corpus', 'prentious': 'pretentious', 'Sussia': 'ancient Jewish village',\n                  'moustachess': 'moustaches', 'Russions': 'Russians', 'Yuguslavia': 'Yugoslavia',\n                  'atrocitties': 'atrocities', 'Muslimophobe': 'Muslim phobic', 'fallicious': 'fallacious',\n                  'recussed': 'recursed', '@ usafmonitor': '', 'lustfly': 'lustful', 'canMuslims': 'can Muslims',\n                  'journalust': 'journalist', 'digustingly': 'disgustingly', 'harasing': 'harassing',\n                  'greatuncle': 'great uncle', 'Drumpf': 'Trump', 'rejectes': 'rejected', 'polyagamous': 'polygamous',\n                  'Mushlims': 'Muslims', 'accusition': 'accusation', 'geniusses': 'geniuses',\n                  'moustachesomething': 'moustache something', 'heineous': 'heinous',\n                  'Sapiosexuals': 'sapiosexual', 'sapiosexuals': 'sapiosexual', 'Sapiosexual': 'sapiosexual',\n                  'sapiosexual': 'Sexually attracted to intelligence', 'pansexuals': 'pansexual',\n                  'autosexual': 'auto sexual', 'sexualSlutty': 'sexual Slutty', 'hetorosexuality': 'hetoro sexuality',\n                  'chinesese': 'chinese', 'pizza gate': 'debunked conspiracy theory',\n                  'countryless': 'Having no country',\n                  'muslimare': 'Muslim are', 'iPhoneX': 'iPhone', 'lionese': 'lioness', 'marionettist': 'Marionettes',\n                  'demonetize': 'demonetized', 'eneyone': 'anyone', 'Karonese': 'Karo people Indonesia',\n                  'minderheid': 'minder worse', 'mainstreamly': 'mainstream', 'contraproductive': 'contra productive',\n                  'diffenky': 'differently', 'abandined': 'abandoned', 'p0 rnstars': 'pornstars',\n                  'overproud': 'over proud',\n                  'cheekboned': 'cheek boned', 'heriones': 'heroines', 'eventhogh': 'even though',\n                  'americanmedicalassoc': 'american medical assoc', 'feelwhen': 'feel when', 'Hhhow': 'how',\n                  'reallySemites': 'really Semites', 'gamergaye': 'gamersgate', 'manspreading': 'man spreading',\n                  'thammana': 'Tamannaah Bhatia', 'dogmans': 'dogmas', 'managementskills': 'management skills',\n                  'mangoliod': 'mongoloid', 'geerymandered': 'gerrymandered', 'mandateing': 'man dateing',\n                  'Romanium': 'Romanum',\n                  'mailwoman': 'mail woman', 'humancoalition': 'human coalition',\n                  'manipullate': 'manipulate', 'everyo0 ne': 'everyone', 'takeove': 'takeover',\n                  'Nonchristians': 'Non Christians', 'goverenments': 'governments', 'govrment': 'government',\n                  'polygomists': 'polygamists', 'Demogorgan': 'Demogorgon', 'maralago': 'Mar-a-Lago',\n                  'antibigots': 'anti bigots', 'gouing': 'going', 'muzaffarbad': 'muzaffarabad',\n                  'suchvstupid': 'such stupid', 'apartheidisrael': 'apartheid israel', \n                  'personaltiles': 'personal titles', 'lawyergirlfriend': 'lawyer girl friend',\n                  'northestern': 'northwestern', 'yeardold': 'years old', 'masskiller': 'mass killer',\n                  'southeners': 'southerners', 'Unitedstatesian': 'United states',\n\n                  'peoplekind': 'people kind', 'peoplelike': 'people like', 'countrypeople': 'country people',\n                  'shitpeople': 'shit people', 'trumpology': 'trump ology', 'trumpites': 'Trump supporters',\n                  'trumplies': 'trump lies', 'donaldtrumping': 'donald trumping', 'trumpdating': 'trump dating',\n                  'trumpsters': 'trumpeters', 'ciswomen': 'cis women', 'womenizer': 'womanizer',\n                  'pregnantwomen': 'pregnant women', 'autoliker': 'auto liker', 'smelllike': 'smell like',\n                  'autolikers': 'auto likers', 'religiouslike': 'religious like', 'likemail': 'like mail',\n                  'fislike': 'dislike', 'sneakerlike': 'sneaker like', 'like⬇': 'like',\n                  'likelovequotes': 'like lovequotes', 'likelogo': 'like logo', 'sexlike': 'sex like',\n                  'Whatwould': 'What would', 'Howwould': 'How would', 'manwould': 'man would',\n                  'exservicemen': 'ex servicemen', 'femenism': 'feminism', 'devopment': 'development',\n                  'doccuments': 'documents', 'supplementplatform': 'supplement platform', 'mendatory': 'mandatory',\n                  'moviments': 'movements', 'Kremenchuh': 'Kremenchug', 'docuements': 'documents',\n                  'determenism': 'determinism', 'envisionment': 'envision ment',\n                  'tricompartmental': 'tri compartmental', 'AddMovement': 'Add Movement',\n                  'mentionong': 'mentioning', 'Whichtreatment': 'Which treatment', 'repyament': 'repayment',\n                  'insemenated': 'inseminated', 'inverstment': 'investment',\n                  'managemental': 'manage mental', 'Inviromental': 'Environmental', 'menstrution': 'menstruation',\n                  'indtrument': 'instrument', 'mentenance': 'maintenance', 'fermentqtion': 'fermentation',\n                  'achivenment': 'achievement', 'mismanagements': 'mis managements', 'requriment': 'requirement',\n                  'denomenator': 'denominator', 'drparment': 'department', 'acumens': 'acumen s',\n                  'celemente': 'Clemente', 'manajement': 'management', 'govermenent': 'government',\n                  'accomplishmments': 'accomplishments', 'rendementry': 'rendement ry',\n                  'repariments': 'departments', 'menstrute': 'menstruate', 'determenistic': 'deterministic',\n                  'resigment': 'resignment', 'selfpayment': 'self payment', 'imrpovement': 'improvement',\n                  'enivironment': 'environment', 'compartmentley': 'compartment',\n                  'augumented': 'augmented', 'parmenent': 'permanent', 'dealignment': 'de alignment',\n                  'develepoments': 'developments', 'menstrated': 'menstruated', 'phnomenon': 'phenomenon',\n                  'Employmment': 'Employment', 'dimensionalise': 'dimensional ise', 'menigioma': 'meningioma',\n                  'recrument': 'recrement', 'Promenient': 'Provenient', 'gonverment': 'government',\n                  'statemment': 'statement', 'recuirement': 'requirement', 'invetsment': 'investment',\n                  'parilment': 'parchment', 'parmently': 'patiently', 'agreementindia': 'agreement india',\n                  'menifesto': 'manifesto', 'accomplsihments': 'accomplishments', 'disangagement': 'disengagement',\n                  'aevelopment': 'development', 'procument': 'procumbent', 'harashment': 'harassment',\n                  'Tiannanmen': 'Tiananmen', 'commensalisms': 'commensal isms', 'devlelpment': 'development',\n                  'dimensons': 'dimensions', 'recruitment2017': 'recruitment 2017', 'polishment': 'pol ishment',\n                  'CommentSafe': 'Comment Safe', 'meausrements': 'measurements', 'geomentrical': 'geometrical',\n                  'undervelopment': 'undevelopment', 'mensurational': 'mensuration al', 'fanmenow': 'fan menow',\n                  'permenganate': 'permanganate', 'bussinessmen': 'businessmen',\n                  'supertournaments': 'super tournaments', 'permanmently': 'permanently',\n                  'lamenectomy': 'lamnectomy', 'assignmentcanyon': 'assignment canyon', 'adgestment': 'adjustment',\n                  'mentalized': 'metalized', 'docyments': 'documents', 'requairment': 'requirement',\n                  'batsmencould': 'batsmen could', 'argumentetc': 'argument etc', 'enjoiment': 'enjoyment',\n                  'invement': 'movement', 'accompliushments': 'accomplishments', 'regements': 'regiments',\n                  'departmentHow': 'department How', 'Aremenian': 'Armenian', 'amenclinics': 'amen clinics',\n                  'nonfermented': 'non fermented', 'Instumentation': 'Instrumentation', 'mentalitiy': 'mentality',\n                  ' govermen ': 'goverment', 'underdevelopement': 'under developement', 'parlimentry': 'parliamentary',\n                  'indemenity': 'indemnity', 'Inatrumentation': 'Instrumentation', 'menedatory': 'mandatory',\n                  'mentiri': 'entire', 'accomploshments': 'accomplishments', 'instrumention': 'instrument ion',\n                  'afvertisements': 'advertisements', 'parlementarian': 'parlement arian',\n                  'entitlments': 'entitlements', 'endrosment': 'endorsement', 'improment': 'impriment',\n                  'archaemenid': 'Achaemenid', 'replecement': 'replacement', 'placdment': 'placement',\n                  'femenise': 'feminise', 'envinment': 'environment', 'AmenityCompany': 'Amenity Company',\n                  'increaments': 'increments', 'accomplihsments': 'accomplishments',\n                  'manygovernment': 'many government', 'panishments': 'punishments', 'elinment': 'eloinment',\n                  'mendalin': 'mend alin', 'farmention': 'farm ention', 'preincrement': 'pre increment',\n                  'postincrement': 'post increment', 'achviements': 'achievements', 'menditory': 'mandatory',\n                  'Emouluments': 'Emoluments', 'Stonemen': 'Stone men', 'menmium': 'medium',\n                  'entaglement': 'entanglement', 'integumen': 'integument', 'harassument': 'harassment',\n                  'retairment': 'retainment', 'enviorement': 'environment', 'tormentous': 'torment ous',\n                  'confiment': 'confident', 'Enchroachment': 'Encroachment', 'prelimenary': 'preliminary',\n                  'fudamental': 'fundamental', 'instrumenot': 'instrument', 'icrement': 'increment',\n                  'prodimently': 'prominently', 'meniss': 'menise', 'Whoimplemented': 'Who implemented',\n                  'Representment': 'Rep resentment', 'StartFragment': 'Start Fragment',\n                  'EndFragment': 'End Fragment', ' documentarie ': ' documentaries ', 'requriments': 'requirements',\n                  'constitutionaldevelopment': 'constitutional development', 'parlamentarians': 'parliamentarians',\n                  'Rumenova': 'Rumen ova', 'argruments': 'arguments', 'findamental': 'fundamental',\n                  'totalinvestment': 'total investment', 'gevernment': 'government', 'recmommend': 'recommend',\n                  'appsmoment': 'apps moment', 'menstruual': 'menstrual', 'immplemented': 'implemented',\n                  'engangement': 'engagement', 'invovement': 'involvement', 'returement': 'retirement',\n                  'simentaneously': 'simultaneously', 'accompishments': 'accomplishments',\n                  'menstraution': 'menstruation', 'experimently': 'experiment', 'abdimen': 'abdomen',\n                  'cemenet': 'cement', 'propelment': 'propel ment', 'unamendable': 'un amendable',\n                  'employmentnews': 'employment news', 'lawforcement': 'law forcement',\n                  'menstuating': 'menstruating', 'fevelopment': 'development', 'reglamented': 'reg lamented',\n                  'imrovment': 'improvement', 'recommening': 'recommending', 'sppliment': 'supplement',\n                  'measument': 'measurement', 'reimbrusement': 'reimbursement', 'Nutrament': 'Nutriment',\n                  'puniahment': 'punishment', 'subligamentous': 'sub ligamentous', 'comlementry': 'complementary',\n                  'reteirement': 'retirement', 'envioronments': 'environments', 'haraasment': 'harassment',\n                  'USAgovernment': 'USA government', 'Apartmentfinder': 'Apartment finder',\n                  'encironment': 'environment', 'metacompartment': 'meta compartment',\n                  'augumentation': 'argumentation', 'dsymenorrhoea': 'dysmenorrhoea',\n                  'nonabandonment': 'non abandonment', 'annoincement': 'announcement',\n                  'menberships': 'memberships', 'Gamenights': 'Game nights', 'enliightenment': 'enlightenment',\n                  'supplymentry': 'supplementary', 'parlamentary': 'parliamentary', 'duramen': 'dura men',\n                  'hotelmanagement': 'hotel management', 'deartment': 'department',\n                  'treatmentshelp': 'treatments help', 'attirements': 'attire ments',\n                  'amendmending': 'amend mending', 'pseudomeningocele': 'pseudo meningocele',\n                  'intrasegmental': 'intra segmental', 'treatmenent': 'treatment', 'infridgement': 'infringement',\n                  'infringiment': 'infringement', 'recrecommend': 'rec recommend', 'entartaiment': 'entertainment',\n                  'inplementing': 'implementing', 'indemendent': 'independent', 'tremendeous': 'tremendous',\n                  'commencial': 'commercial', 'scomplishments': 'accomplishments', 'Emplement': 'Implement',\n                  'dimensiondimensions': 'dimension dimensions', 'depolyment': 'deployment',\n                  'conpartment': 'compartment', 'govnments': 'movements', 'menstrat': 'menstruate',\n                  'accompplishments': 'accomplishments', 'Enchacement': 'Enchancement',\n                  'developmenent': 'development', 'emmenagogues': 'emmenagogue', 'aggeement': 'agreement',\n                  'elementsbond': 'elements bond', 'remenant': 'remnant', 'Manamement': 'Management',\n                  'Augumented': 'Augmented', 'dimensonless': 'dimensionless',\n                  'ointmentsointments': 'ointments ointments', 'achiements': 'achievements',\n                  'recurtment': 'recurrent', 'gouverments': 'governments', 'docoment': 'document',\n                  'programmingassignments': 'programming assignments', 'menifest': 'manifest',\n                  'investmentguru': 'investment guru', 'deployements': 'deployments', 'Invetsment': 'Investment',\n                  'plaement': 'placement', 'Perliament': 'Parliament', 'femenists': 'feminists',\n                  'ecumencial': 'ecumenical', 'advamcements': 'advancements', 'refundment': 'refund ment',\n                  'settlementtake': 'settlement take', 'mensrooms': 'mens rooms',\n                  'productManagement': 'product Management', 'armenains': 'armenians',\n                  'betweenmanagement': 'between management', 'difigurement': 'disfigurement',\n                  'Armenized': 'Armenize', 'hurrasement': 'hurra sement', 'mamgement': 'management',\n                  'momuments': 'monuments', 'eauipments': 'equipments', 'managemenet': 'management',\n                  'treetment': 'treatment', 'webdevelopement': 'web developement', 'supplemenary': 'supplementary',\n                  'Encironmental': 'Environmental', 'Understandment': 'Understand ment',\n                  'enrollnment': 'enrollment', 'thinkstrategic': 'think strategic', 'thinkinh': 'thinking',\n                  'Softthinks': 'Soft thinks', 'underthinking': 'under thinking', 'thinksurvey': 'think survey',\n                  'whitelash': 'white lash', 'whiteheds': 'whiteheads', 'whitetning': 'whitening',\n                  'whitegirls': 'white girls', 'whitewalkers': 'white walkers', 'manycountries': 'many countries',\n                  'accomany': 'accompany', 'fromGermany': 'from Germany', 'manychat': 'many chat',\n                  'Germanyl': 'Germany l', 'manyness': 'many ness', 'many4': 'many', 'exmuslims': 'ex muslims',\n                  'digitizeindia': 'digitize india', 'indiarush': 'india rush', 'indiareads': 'india reads',\n                  'telegraphindia': 'telegraph india', 'Southindia': 'South india', 'Airindia': 'Air india',\n                  'siliconindia': 'silicon india', 'airindia': 'air india', 'indianleaders': 'indian leaders',\n                  'fundsindia': 'funds india', 'indianarmy': 'indian army', 'Technoindia': 'Techno india',\n                  'Betterindia': 'Better india', 'capesindia': 'capes india', 'Rigetti': 'Ligetti',\n                  'vegetablr': 'vegetable', 'get90': 'get', 'Magetta': 'Maretta', 'nagetive': 'native',\n                  'isUnforgettable': 'is Unforgettable', 'get630': 'get 630', 'GadgetPack': 'Gadget Pack',\n                  'Languagetool': 'Language tool', 'bugdget': 'budget', 'africaget': 'africa get',\n                  'ABnegetive': 'Abnegative', 'orangetheory': 'orange theory', 'getsmuggled': 'get smuggled',\n                  'avegeta': 'ave geta', 'gettubg': 'getting', 'gadgetsnow': 'gadgets now',\n                  'surgetank': 'surge tank', 'gadagets': 'gadgets', 'getallparts': 'get allparts',\n                  'messenget': 'messenger', 'vegetarean': 'vegetarian', 'get1000': 'get 1000',\n                  'getfinancing': 'get financing', 'getdrip': 'get drip', 'AdsTargets': 'Ads Targets',\n                  'tgethr': 'together', 'vegetaries': 'vegetables', 'forgetfulnes': 'forgetfulness',\n                  'fisgeting': 'fidgeting', 'BudgetAir': 'Budget Air',\n                  'getDepersonalization': 'get Depersonalization', 'negetively': 'negatively',\n                  'gettibg': 'getting', 'nauget': 'naught', 'Bugetti': 'Bugatti', 'plagetum': 'plage tum',\n                  'vegetabale': 'vegetable', 'changetip': 'change tip', 'blackwashing': 'black washing',\n                  'blackpink': 'black pink', 'blackmoney': 'black money',\n                  'blackmarks': 'black marks', 'blackbeauty': 'black beauty', 'unblacklisted': 'un blacklisted',\n                  'blackdotes': 'black dotes', 'blackboxing': 'black boxing', 'blackpaper': 'black paper',\n                  'blackpower': 'black power', 'Latinamericans': 'Latin americans', 'musigma': 'mu sigma',\n                  'Indominus': 'In dominus', 'usict': 'USSCt', 'indominus': 'in dominus', 'Musigma': 'Mu sigma',\n                  'plus5': 'plus', 'Russiagate': 'Russia gate', 'russophobic': 'Russophobiac',\n                  'Marcusean': 'Marcuse an', 'Radijus': 'Radius', 'cobustion': 'combustion',\n                  'Austrialians': 'Australians', 'mylogenous': 'myogenous', 'Raddus': 'Radius',\n                  'hetrogenous': 'heterogenous', 'greenhouseeffect': 'greenhouse effect', 'aquous': 'aqueous',\n                  'Taharrush': 'Tahar rush', 'Senousa': 'Venous', 'diplococcus': 'diplo coccus',\n                  'CityAirbus': 'City Airbus', 'sponteneously': 'spontaneously', 'trustless': 't rustless',\n                  'Pushkaram': 'Pushkara m', 'Fusanosuke': 'Fu sanosuke', 'isthmuses': 'isthmus es',\n                  'lucideus': 'lucidum', 'overjustification': 'over justification', 'Bindusar': 'Bind usar',\n                  'cousera': 'couler', 'musturbation': 'masturbation', 'infustry': 'industry',\n                  'Huswifery': 'Huswife ry', 'rombous': 'bombous', 'disengenuously': 'disingenuously',\n                  'sllybus': 'syllabus', 'celcious': 'delicious', 'cellsius': 'celsius',\n                  'lethocerus': 'Lethocerus', 'monogmous': 'monogamous', 'Ballyrumpus': 'Bally rumpus',\n                  'Koushika': 'Koushik a', 'vivipoarous': 'viviparous', 'ludiculous': 'ridiculous',\n                  'sychronous': 'synchronous', 'industiry': 'industry', 'scuduse': 'scud use',\n                  'babymust': 'baby must', 'simultqneously': 'simultaneously', 'exust': 'ex ust',\n                  'notmusing': 'not musing', 'Zamusu': 'Amuse', 'tusaki': 'tu saki', 'Marrakush': 'Marrakesh',\n                  'justcheaptickets': 'just cheaptickets', 'Ayahusca': 'Ayahausca', 'samousa': 'samosa',\n                  'Gusenberg': 'Gutenberg', 'illustratuons': 'illustrations', 'extemporeneous': 'extemporaneous',\n                  'Mathusla': 'Mathusala', 'Confundus': 'Con fundus', 'tusts': 'trusts', 'poisenious': 'poisonous',\n                  'Mevius': 'Medius', 'inuslating': 'insulating', 'aroused21000': 'aroused 21000',\n                  'Wenzeslaus': 'Wenceslaus', 'JustinKase': 'Justin Kase', 'purushottampur': 'purushottam pur',\n                  'citruspay': 'citrus pay', 'secutus': 'sects', 'austentic': 'austenitic',\n                  'FacePlusPlus': 'Face PlusPlus', 'aysnchronous': 'asynchronous',\n                  'teamtreehouse': 'team treehouse', 'uncouncious': 'unconscious', 'Priebuss': 'Prie buss',\n                  'consciousuness': 'consciousness', 'susubsoil': 'su subsoil', 'trimegistus': 'Trismegistus',\n                  'protopeterous': 'protopterous', 'trustworhty': 'trustworthy', 'ushually': 'usually',\n                  'industris': 'industries', 'instantneous': 'instantaneous', 'superplus': 'super plus',\n                  'shrusti': 'shruti', 'hindhus': 'hindus', 'outonomous': 'autonomous', 'reliegious': 'religious',\n                  'Kousakis': 'Kou sakis', 'reusult': 'result', 'JanusGraph': 'Janus Graph',\n                  'palusami': 'palus ami', 'mussraff': 'muss raff', 'hukous': 'humous',\n                  'photoacoustics': 'photo acoustics', 'kushanas': 'kusha nas', 'justdile': 'justice',\n                  'Massahusetts': 'Massachusetts', 'uspset': 'upset', 'sustinet': 'sustinent',\n                  'consicious': 'conscious', 'Sadhgurus': 'Sadh gurus', 'hystericus': 'hysteric us',\n                  'visahouse': 'visa house', 'supersynchronous': 'super synchronous', 'posinous': 'rosinous',\n                  'Fernbus': 'Fern bus', 'Tiltbrush': 'Tilt brush', 'glueteus': 'gluteus', 'posionus': 'poisons',\n                  'Freus': 'Frees', 'Zhuchengtyrannus': 'Zhucheng tyrannus', 'savonious': 'sanious',\n                  'CusJo': 'Cusco', 'congusion': 'confusion', 'dejavus': 'dejavu s', 'uncosious': 'uncopious',\n                  'previius': 'previous', 'counciousness': 'conciousness', 'lustorus': 'lustrous',\n                  'sllyabus': 'syllabus', 'mousquitoes': 'mosquitoes', 'Savvius': 'Savvies', 'arceius': 'Arcesius',\n                  'prejusticed': 'prejudiced', 'requsitioned': 'requisitioned',\n                  'deindustralization': 'deindustrialization', 'muscleblaze': 'muscle blaze',\n                  'ConsciousX5': 'conscious', 'nitrogenious': 'nitrogenous', 'mauritious': 'mauritius',\n                  'rigrously': 'rigorously', 'Yutyrannus': 'Yu tyrannus', 'muscualr': 'muscular',\n                  'conscoiusness': 'consciousness', 'Causians': 'Crusians', 'WorkFusion': 'Work Fusion',\n                  'puspak': 'pu spak', 'Inspirus': 'Inspires', 'illiustrations': 'illustrations',\n                  'Nobushi': 'No bushi', 'theuseof': 'thereof', 'suspicius': 'suspicious', 'Intuous': 'Virtuous',\n                  'gaushalas': 'gaus halas', 'campusthrough': 'campus through', 'seriousity': 'seriosity',\n                  'resustence': 'resistence', 'geminatus': 'geminates', 'disquss': 'discuss',\n                  'nicholus': 'nicholas', 'Husnai': 'Hussar', 'diiscuss': 'discuss', 'diffussion': 'diffusion',\n                  'phusicist': 'physicist', 'ernomous': 'enormous', 'Khushali': 'Khushal i', 'heitus': 'Leitus',\n                  'cracksbecause': 'cracks because', 'Nautlius': 'Nautilus', 'trausted': 'trusted',\n                  'Dardandus': 'Dardanus', 'Megatapirus': 'Mega tapirus', 'clusture': 'culture',\n                  'vairamuthus': 'vairamuthu s', 'disclousre': 'disclosure',\n                  'industrilaization': 'industrialization', 'musilms': 'muslims', 'Australia9': 'Australian',\n                  'causinng': 'causing', 'ibdustries': 'industries', 'searious': 'serious',\n                  'Coolmuster': 'Cool muster', 'sissyphus': 'sisyphus', ' justificatio ': 'justification',\n                  'antihindus': 'anti hindus', 'Moduslink': 'Modus link', 'zymogenous': 'zymogen ous',\n                  'prospeorus': 'prosperous', 'Retrocausality': 'Retro causality', 'FusionGPS': 'Fusion GPS',\n                  'Mouseflow': 'Mouse flow', 'bootyplus': 'booty plus', 'Itylus': 'I tylus',\n                  'Olnhausen': 'Olshausen', 'suspeect': 'suspect', 'entusiasta': 'enthusiast',\n                  'fecetious': 'facetious', 'bussiest': 'fussiest', 'Draconius': 'Draconis',\n                  'requsite': 'requisite', 'nauseatic': 'nausea tic', 'Brusssels': 'Brussels',\n                  'repurcussion': 'repercussion', 'Jeisus': 'Jesus', 'philanderous': 'philander ous',\n                  'muslisms': 'muslims', 'august2017': 'august 2017', 'calccalculus': 'calc calculus',\n                  'unanonymously': 'un anonymously', 'Imaprtus': 'Impetus', 'carnivorus': 'carnivorous',\n                  'Corypheus': 'Coryphees', 'austronauts': 'astronauts', 'neucleus': 'nucleus',\n                  'housepoor': 'house poor', 'rescouses': 'responses', 'Tagushi': 'Tagus hi',\n                  'hyperfocusing': 'hyper focusing', 'nutriteous': 'nutritious', 'chylus': 'chylous',\n                  'preussure': 'pressure', 'outfocus': 'out focus', 'Hanfus': 'Hannus', 'Rustyrose': 'Rusty rose',\n                  'vibhushant': 'vibhushan t', 'conciousnes': 'conciousness', 'Venus25': 'Venus',\n                  'Sedataious': 'Seditious', 'promuslim': 'pro muslim', 'statusGuru': 'status Guru',\n                  'yousician': 'musician', 'transgenus': 'trans genus', 'Pushbullet': 'Push bullet',\n                  'jeesyllabus': 'jee syllabus', 'complusary': 'compulsory', 'Holocoust': 'Holocaust',\n                  'careerplus': 'career plus', 'Lllustrate': 'Illustrate', 'Musino': 'Musion',\n                  'Phinneus': 'Phineus', 'usedtoo': 'used too', 'JustBasic': 'Just Basic', 'webmusic': 'web music',\n                  'TrustKit': 'Trust Kit', 'industrZgies': 'industries', 'rubustness': 'robustness',\n                  'Missuses': 'Miss uses', 'Musturbation': 'Masturbation', 'bustees': 'bus tees',\n                  'justyfy': 'justify', 'pegusus': 'pegasus', 'industrybuying': 'industry buying',\n                  'advantegeous': 'advantageous', 'kotatsus': 'kotatsu s', 'justcreated': 'just created',\n                  'simultameously': 'simultaneously', 'husoone': 'huso one', 'twiceusing': 'twice using',\n                  'cetusplay': 'cetus play', 'sqamous': 'squamous', 'claustophobic': 'claustrophobic',\n                  'Kaushika': 'Kaushik a', 'dioestrus': 'di oestrus', 'Degenerous': 'De generous',\n                  'neculeus': 'nucleus', 'cutaneously': 'cu taneously', 'Alamotyrannus': 'Alamo tyrannus',\n                  'Ivanious': 'Avanious', 'arceous': 'araceous', 'Flixbus': 'Flix bus', 'caausing': 'causing',\n                  'publious': 'Publius', 'Juilus': 'Julius', 'Australianism': 'Australian ism',\n                  'vetronus': 'verrons', 'nonspontaneous': 'non spontaneous', 'calcalus': 'calculus',\n                  'commudus': 'Commodus', 'Rheusus': 'Rhesus', 'syallubus': 'syllabus', 'Yousician': 'Musician',\n                  'qurush': 'qu rush', 'athiust': 'athirst', 'conclusionless': 'conclusion less',\n                  'usertesting': 'user testing', 'redius': 'radius', 'Austrolia': 'Australia',\n                  'sllaybus': 'syllabus', 'toponymous': 'top onymous', 'businiss': 'business',\n                  'hyperthalamus': 'hyper thalamus', 'clause55': 'clause', 'cosicous': 'conscious',\n                  'Sushena': 'Saphena', 'Luscinus': 'Luscious', 'Prussophile': 'Russophile', 'jeaslous': 'jealous',\n                  'Austrelia': 'Australia', 'contiguious': 'contiguous',\n                  'subconsciousnesses': 'sub consciousnesses', ' jusification ': 'justification',\n                  'dilusion': 'delusion', 'anticoncussive': 'anti concussive', 'disngush': 'disgust',\n                  'constiously': 'consciously', 'filabustering': 'filibustering', 'GAPbuster': 'GAP buster',\n                  'insectivourous': 'insectivorous', 'glocuse': 'louse', 'Antritrust': 'Antitrust',\n                  'thisAustralian': 'this Australian', 'FusionDrive': 'Fusion Drive', 'nuclus': 'nucleus',\n                  'abussive': 'abusive', 'mustang1': 'mustangs', 'inradius': 'in radius', 'polonious': 'polonius',\n                  'ofKulbhushan': 'of Kulbhushan', 'homosporous': 'homos porous', 'circumradius': 'circum radius',\n                  'atlous': 'atrous', 'insustry': 'industry', 'campuswith': 'campus with', 'beacsuse': 'because',\n                  'concuous': 'conscious', 'nonHindus': 'non Hindus', 'carnivourous': 'carnivorous',\n                  'tradeplus': 'trade plus', 'Jeruselam': 'Jerusalem',\n                  'musuclar': 'muscular', 'deangerous': 'dangerous', 'disscused': 'discussed',\n                  'industdial': 'industrial', 'sallatious': 'fallacious', 'rohmbus': 'rhombus',\n                  'golusu': 'gol usu', 'Minangkabaus': 'Minangkabau s', 'Mustansiriyah': 'Mustansiriya h',\n                  'anomymously': 'anonymously', 'abonymously': 'anonymously', 'indrustry': 'industry',\n                  'Musharrf': 'Musharraf', 'workouses': 'workhouses', 'sponataneously': 'spontaneously',\n                  'anmuslim': 'an muslim', 'syallbus': 'syllabus', 'presumptuousnes': 'presumptuousness',\n                  'Thaedus': 'Thaddus', 'industey': 'industry', 'hkust': 'hust', 'Kousseri': 'Kousser i',\n                  'mousestats': 'mouses tats', 'russiagate': 'russia gate', 'simantaneously': 'simultaneously',\n                  'Austertana': 'Auster tana', 'infussions': 'infusions', 'coclusion': 'conclusion',\n                  'sustainabke': 'sustainable', 'tusami': 'tu sami', 'anonimously': 'anonymously',\n                  'usebase': 'use base', 'balanoglossus': 'Balanoglossus', 'Unglaus': 'Ung laus',\n                  'ignoramouses': 'ignoramuses', 'snuus': 'snugs', 'reusibility': 'reusability',\n                  'Straussianism': 'Straussian ism', 'simoultaneously': 'simultaneously',\n                  'realbonus': 'real bonus', 'nuchakus': 'nunchakus', 'annonimous': 'anonymous',\n                  'Incestious': 'Incestuous', 'Manuscriptology': 'Manuscript ology', 'difusse': 'diffuse',\n                  'Pliosaurus': 'Pliosaur us', 'cushelle': 'cush elle', 'Catallus': 'Catullus',\n                  'MuscleBlaze': 'Muscle Blaze', 'confousing': 'confusing', 'enthusiasmless': 'enthusiasm less',\n                  'Tetherusd': 'Tethered', 'Josephius': 'Josephus', 'jusrlt': 'just',\n                  'simutaneusly': 'simultaneously', 'mountaneous': 'mountainous', 'Badonicus': 'Sardonicus',\n                  'muccus': 'mucous', 'nicus': 'nidus', 'austinlizards': 'austin lizards',\n                  'errounously': 'erroneously', 'Australua': 'Australia', 'sylaabus': 'syllabus',\n                  'dusyant': 'distant', 'javadiscussion': 'java discussion', 'megabuses': 'mega buses',\n                  'danergous': 'dangerous', 'contestious': 'contentious', 'exause': 'excuse',\n                  'muscluar': 'muscular', 'avacous': 'vacuous', 'Ingenhousz': 'Ingenious',\n                  'holocausting': 'holocaust ing', 'Pakustan': 'Pakistan', 'purusharthas': 'purushartha',\n                  'bapus': 'bapu s', 'useul': 'useful', 'pretenious': 'pretentious', 'homogeneus': 'homogeneous',\n                  'bhlushes': 'blushes', 'Saggittarius': 'Sagittarius', 'sportsusa': 'sports usa',\n                  'kerataconus': 'keratoconus', 'infrctuous': 'infectuous', 'Anonoymous': 'Anonymous',\n                  'triphosphorus': 'tri phosphorus', 'ridicjlously': 'ridiculously',\n                  'worldbusiness': 'world business', 'hollcaust': 'holocaust', 'Dusra': 'Dura',\n                  'meritious': 'meritorious', 'Sauskes': 'Causes', 'inudustry': 'industry',\n                  'frustratd': 'frustrate', 'hypotenous': 'hypogenous', 'Dushasana': 'Dush asana',\n                  'saadus': 'status', 'keratokonus': 'keratoconus', 'Jarrus': 'Harrus', 'neuseous': 'nauseous',\n                  'simutanously': 'simultaneously', 'diphosphorus': 'di phosphorus', 'sulprus': 'surplus',\n                  'Hasidus': 'Hasid us', 'suspenive': 'suspensive', 'illlustrator': 'illustrator',\n                  'userflows': 'user flows', 'intrusivethoughts': 'intrusive thoughts', 'countinous': 'continuous',\n                  'gpusli': 'gusli', 'Calculus1': 'Calculus', 'bushiri': 'Bushire',\n                  'torvosaurus': 'Torosaurus', 'chestbusters': 'chest busters', 'Satannus': 'Sat annus',\n                  'falaxious': 'fallacious', 'obnxious': 'obnoxious', 'tranfusions': 'transfusions',\n                  'PlayMagnus': 'Play Magnus', 'Epicodus': 'Episodes', 'Hypercubus': 'Hypercubes',\n                  'Musickers': 'Musick ers', 'programmebecause': 'programme because', 'indiginious': 'indigenous',\n                  'housban': 'Housman', 'iusso': 'kusso', 'annilingus': 'anilingus', 'Nennus': 'Genius',\n                  'pussboy': 'puss boy', 'Photoacoustics': 'Photo acoustics', 'Hindusthanis': 'Hindustanis',\n                  'lndustrial': 'industrial', 'tyrannously': 'tyrannous', 'Susanoomon': 'Susanoo mon',\n                  'colmbus': 'columbus', 'sussessful': 'successful', 'ousmania': 'ous mania',\n                  'ilustrating': 'illustrating', 'famousbirthdays': 'famous birthdays',\n                  'suspectance': 'suspect ance', 'extroneous': 'extraneous', 'teethbrush': 'teeth brush',\n                  'abcmouse': 'abc mouse', 'degenerous': 'de generous', 'doesGauss': 'does Gauss',\n                  'insipudus': 'insipidus', 'movielush': 'movie lush', 'Rustichello': 'Rustic hello',\n                  'Firdausiya': 'Firdausi ya', 'checkusers': 'check users', 'householdware': 'household ware',\n                  'prosporously': 'prosperously', 'SteLouse': 'Ste Louse', 'obfuscaton': 'obfuscation',\n                  'amorphus': 'amorph us', 'trustworhy': 'trustworthy', 'celsious': 'cesious',\n                  'dangorous': 'dangerous', 'anticancerous': 'anti cancerous', 'cousi ': 'cousin ',\n                  'austroloid': 'australoid', 'fergussion': 'percussion', 'andKyokushin': 'and Kyokushin',\n                  'cousan': 'cousin', 'Huskystar': 'Hu skystar', 'retrovisus': 'retrovirus', 'becausr': 'because',\n                  'Jerusalsem': 'Jerusalem', 'motorious': 'notorious', 'industrilised': 'industrialised',\n                  'powerballsusa': 'powerballs usa', 'monoceious': 'monoecious', 'batteriesplus': 'batteries plus',\n                  'nonviscuous': 'nonviscous', 'industion': 'induction', 'bussinss': 'bussings',\n                  'userbags': 'user bags', 'Jlius': 'Julius', 'thausand': 'thousand', 'plustwo': 'plus two',\n                  'defpush': 'def push', 'subconcussive': 'sub concussive', 'muslium': 'muslim',\n                  'industrilization': 'industrialization', 'Maurititus': 'Mauritius', 'uslme': 'some',\n                  'Susgaon': 'Surgeon', 'Pantherous': 'Panther ous', 'antivirius': 'antivirus',\n                  'Trustclix': 'Trust clix', 'silumtaneously': 'simultaneously', 'Icompus': 'Corpus',\n                  'atonomous': 'autonomous', 'Reveuse': 'Reve use', 'legumnous': 'leguminous',\n                  'syllaybus': 'syllabus', 'louspeaker': 'loudspeaker', 'susbtraction': 'substraction',\n                  'virituous': 'virtuous', 'disastrius': 'disastrous', 'jerussalem': 'jerusalem',\n                  'Industrailzed': 'Industrialized', 'recusion': 'recushion',\n                  'simultenously': 'simultaneously',\n                  'Pulphus': 'Pulpous', 'harbaceous': 'herbaceous', 'phlegmonous': 'phlegmon ous', 'use38': 'use',\n                  'jusify': 'justify', 'instatanously': 'instantaneously', 'tetramerous': 'tetramer ous',\n                  'usedvin': 'used vin', 'sagittarious': 'sagittarius', 'mausturbate': 'masturbate',\n                  'subcautaneous': 'subcutaneous', 'dangergrous': 'dangerous', 'sylabbus': 'syllabus',\n                  'hetorozygous': 'heterozygous', 'Ignasius': 'Ignacius', 'businessbor': 'business bor',\n                  'Bhushi': 'Thushi', 'Moussolini': 'Mussolini', 'usucaption': 'usu caption',\n                  'Customzation': 'Customization', 'cretinously': 'cretinous', 'genuiuses': 'geniuses',\n                  'Moushmee': 'Mousmee', 'neigous': 'nervous',\n                  'infrustructre': 'infrastructure', 'Ilusha': 'Ilesha', 'suconciously': 'unconciously',\n                  'stusy': 'study', 'mustectomy': 'mastectomy', 'Farmhousebistro': 'Farmhouse bistro',\n                  'instantanous': 'instantaneous', 'JustForex': 'Just Forex', 'Indusyry': 'Industry',\n                  'mustabating': 'must abating', 'uninstrusive': 'unintrusive', 'customshoes': 'customs hoes',\n                  'homageneous': 'homogeneous', 'Empericus': 'Imperious', 'demisexuality': 'demi sexuality',\n                  'transexualism': 'transsexualism', 'sexualises': 'sexualise', 'demisexuals': 'demisexual',\n                  'sexuly': 'sexily', 'Pornosexuality': 'Porno sexuality', 'sexond': 'second', 'sexxual': 'sexual',\n                  'asexaul': 'asexual', 'sextactic': 'sex tactic', 'sexualityism': 'sexuality ism',\n                  'monosexuality': 'mono sexuality', 'intwrsex': 'intersex', 'hypersexualize': 'hyper sexualize',\n                  'homosexualtiy': 'homosexuality', 'examsexams': 'exams exams', 'sexmates': 'sex mates',\n                  'sexyjobs': 'sexy jobs', 'sexitest': 'sexiest', 'fraysexual': 'fray sexual',\n                  'sexsurrogates': 'sex surrogates', 'sexuallly': 'sexually', 'gamersexual': 'gamer sexual',\n                  'greysexual': 'grey sexual', 'omnisexuality': 'omni sexuality', 'hetereosexual': 'heterosexual',\n                  'productsexamples': 'products examples', 'sexgods': 'sex gods', 'semisexual': 'semi sexual',\n                  'homosexulity': 'homosexuality', 'sexeverytime': 'sex everytime', 'neurosexist': 'neuro sexist',\n                  'worldquant': 'world quant', 'Freshersworld': 'Freshers world', 'smartworld': 'sm artworld',\n                  'Mistworlds': 'Mist worlds', 'boothworld': 'booth world', 'ecoworld': 'eco world',\n                  'Ecoworld': 'Eco world', 'underworldly': 'under worldly', 'worldrank': 'world rank',\n                  'Clearworld': 'Clear world', 'Boothworld': 'Booth world', 'Rimworld': 'Rim world',\n                  'cryptoworld': 'crypto world', 'machineworld': 'machine world', 'worldwideley': 'worldwide ley',\n                  'capuletwant': 'capulet want', 'Bhagwanti': 'Bhagwant i', 'Unwanted72': 'Unwanted 72',\n                  'wantrank': 'want rank',\n                  'willhappen': 'will happen', 'thateasily': 'that easily',\n                  'Whatevidence': 'What evidence', 'metaphosphates': 'meta phosphates',\n                  'exilarchate': 'exilarch ate', 'aulphate': 'sulphate', 'Whateducation': 'What education',\n                  'persulphates': 'per sulphates', 'disulphate': 'di sulphate', 'picosulphate': 'pico sulphate',\n                  'tetraosulphate': 'tetrao sulphate', 'prechinese': 'pre chinese',\n                  'Hellochinese': 'Hello chinese', 'muchdeveloped': 'much developed', 'stomuch': 'stomach',\n                  'Whatmakes': 'What makes', 'Lensmaker': 'Lens maker', 'eyemake': 'eye make',\n                  'Techmakers': 'Tech makers', 'cakemaker': 'cake maker', 'makeup411': 'makeup 411',\n                  'objectmake': 'object make', 'crazymaker': 'crazy maker', 'techmakers': 'tech makers',\n                  'makedonian': 'macedonian', 'makeschool': 'make school', 'anxietymake': 'anxiety make',\n                  'makeshifter': 'make shifter', 'countryball': 'country ball', 'Whichcountry': 'Which country',\n                  'countryHow': 'country How', 'Zenfone': 'Zen fone', 'Electroneum': 'Electro neum',\n                  'electroneum': 'electro neum', 'Demonetisation': 'demonetization', 'zenfone': 'zen fone',\n                  'ZenFone': 'Zen Fone', 'onecoin': 'one coin', 'demonetizing': 'demonetized',\n                  'iphone7': 'iPhone', 'iPhone6': 'iPhone', 'microneedling': 'micro needling', 'iphone6': 'iPhone',\n                  'Monegasques': 'Monegasque s', 'demonetised': 'demonetized',\n                  'EveryoneDiesTM': 'EveryoneDies TM', 'teststerone': 'testosterone', 'DoneDone': 'Done Done',\n                  'papermoney': 'paper money', 'Sasabone': 'Sasa bone', 'Blackphone': 'Black phone',\n                  'Bonechiller': 'Bone chiller', 'Moneyfront': 'Money front', 'workdone': 'work done',\n                  'iphoneX': 'iPhone', 'roxycodone': 'r oxycodone',\n                  'moneycard': 'money card', 'Fantocone': 'Fantocine', 'eletronegativity': 'electronegativity',\n                  'mellophones': 'mellophone s', 'isotones': 'iso tones', 'donesnt': 'doesnt',\n                  'thereanyone': 'there anyone', 'electronegativty': 'electronegativity',\n                  'commissiioned': 'commissioned', 'earvphone': 'earphone', 'condtioners': 'conditioners',\n                  'demonetistaion': 'demonetization', 'ballonets': 'ballo nets', 'DoneClaim': 'Done Claim',\n                  'alimoney': 'alimony', 'iodopovidone': 'iodo povidone', 'bonesetters': 'bone setters',\n                  'componendo': 'compon endo', 'probationees': 'probationers', 'one300': 'one 300',\n                  'nonelectrolyte': 'non electrolyte', 'ozonedepletion': 'ozone depletion',\n                  'Stonehart': 'Stone hart', 'Vodafone2': 'Vodafones', 'chaparone': 'chaperone',\n                  'Noonein': 'Noo nein', 'Frosione': 'Erosion', 'IPhone7': 'Iphone', 'pentanone': 'penta none',\n                  'poneglyphs': 'pone glyphs', 'cyclohexenone': 'cyclohexanone', 'marlstone': 'marls tone',\n                  'androneda': 'andromeda', 'iphone8': 'iPhone', 'acidtone': 'acid tone',\n                  'noneconomically': 'non economically', 'Honeyfund': 'Honey fund', 'germanophone': 'Germanophobe',\n                  'Democratizationed': 'Democratization ed', 'haoneymoon': 'honeymoon', 'iPhone7': 'iPhone 7',\n                  'someonewith': 'some onewith', 'Hexanone': 'Hexa none', 'bonespur': 'bones pur',\n                  'sisterzoned': 'sister zoned', 'HasAnyone': 'Has Anyone',\n                  'stonepelters': 'stone pelters', 'Chronexia': 'Chronaxia', 'brotherzone': 'brother zone',\n                  'brotherzoned': 'brother zoned', 'fonecare': 'f onecare', 'nonexsistence': 'nonexistence',\n                  'conents': 'contents', 'phonecases': 'phone cases', 'Commissionerates': 'Commissioner ates',\n                  'activemoney': 'active money', 'dingtone': 'ding tone', 'wheatestone': 'wheatstone',\n                  'chiropractorone': 'chiropractor one', 'heeadphones': 'headphones', 'Maimonedes': 'Maimonides',\n                  'onepiecedeals': 'onepiece deals', 'oneblade': 'one blade', 'venetioned': 'Venetianed',\n                  'sunnyleone': 'sunny leone', 'prendisone': 'prednisone', 'Anglosaxophone': 'Anglo saxophone',\n                  'Blackphones': 'Black phones', 'jionee': 'jinnee', 'chromonema': 'chromo nema',\n                  'iodoketones': 'iodo ketones', 'demonetizations': 'demonetization', 'aomeone': 'someone',\n                  'trillonere': 'trillones', 'abandonee': 'abandon',\n                  'MasterColonel': 'Master Colonel', 'fronend': 'friend', 'Wildstone': 'Wilds tone',\n                  'patitioned': 'petitioned', 'lonewolfs': 'lone wolfs', 'Spectrastone': 'Spectra stone',\n                  'dishonerable': 'dishonorable', 'poisiones': 'poisons',\n                  'condioner': 'conditioner', 'unpermissioned': 'unper missioned', 'friedzone': 'fried zone',\n                  'umumoney': 'umu money', 'anyonestudied': 'anyone studied', 'dictioneries': 'dictionaries',\n                  'nosebone': 'nose bone', 'ofVodafone': 'of Vodafone',\n                  'Yumstone': 'Yum stone', 'oxandrolonesteroid': 'oxandrolone steroid',\n                  'Mifeprostone': 'Mifepristone', 'pheramones': 'pheromones',\n                  'sinophone': 'Sinophobe', 'peloponesian': 'peloponnesian', 'michrophone': 'microphone',\n                  'commissionets': 'commissioners', 'methedone': 'methadone', 'cobditioners': 'conditioners',\n                  'urotone': 'protone', 'smarthpone': 'smartphone', 'conecTU': 'connect you', 'beloney': 'boloney',\n                  'comfortzone': 'comfort zone', 'testostersone': 'testosterone', 'camponente': 'component',\n                  'Idonesia': 'Indonesia', 'dolostones': 'dolostone', 'psiphone': 'psi phone',\n                  'ceftriazone': 'ceftriaxone', 'feelonely': 'feel onely', 'monetation': 'moderation',\n                  'activationenergy': 'activation energy', 'moneydriven': 'money driven',\n                  'staionery': 'stationery', 'zoneflex': 'zone flex', 'moneycash': 'money cash',\n                  'conectiin': 'connection', 'Wannaone': 'Wanna one',\n                  'Pictones': 'Pict ones', 'demonentization': 'demonetization',\n                  'phenonenon': 'phenomenon', 'evenafter': 'even after', 'Sevenfriday': 'Seven friday',\n                  'Devendale': 'Evendale', 'theeventchronicle': 'the event chronicle',\n                  'seventysomething': 'seventy something', 'sevenpointed': 'seven pointed',\n                  'richfeel': 'rich feel', 'overfeel': 'over feel', 'feelingstupid': 'feeling stupid',\n                  'Photofeeler': 'Photo feeler', 'feelomgs': 'feelings', 'feelinfs': 'feelings',\n                  'PlayerUnknown': 'Player Unknown', 'Playerunknown': 'Player unknown', 'knowlefge': 'knowledge',\n                  'knowledgd': 'knowledge', 'knowledeg': 'knowledge', 'knowble': 'Knowle', 'Howknow': 'Howk now',\n                  'knowledgeWoods': 'knowledge Woods', 'knownprogramming': 'known programming',\n                  'selfknowledge': 'self knowledge', 'knowldage': 'knowledge', 'knowyouve': 'know youve',\n                  'aknowlege': 'knowledge', 'Audetteknown': 'Audette known', 'knowlegdeable': 'knowledgeable',\n                  'trueoutside': 'true outside', 'saynthesize': 'synthesize', 'EssayTyper': 'Essay Typer',\n                  'meesaya': 'mee saya', 'Rasayanam': 'Rasayan am', 'fanessay': 'fan essay', 'momsays': 'moms ays',\n                  'sayying': 'saying', 'saydaw': 'say daw', 'Fanessay': 'Fan essay', 'theyreally': 'they really',\n                  'gayifying': 'gayed up with homosexual love', 'gayke': 'gay Online retailers',\n                  'Lingayatism': 'Lingayat',\n                  'macapugay': 'Macaulay', 'jewsplain': 'jews plain',\n                  'banggood': 'bang good', 'goodfriends': 'good friends',\n                  'goodfirms': 'good firms', 'Banggood': 'Bang good', 'dogooder': 'do gooder',\n                  'stillshots': 'stills hots', 'stillsuits': 'still suits', 'panromantic': 'pan romantic',\n                  'paracommando': 'para commando', 'romantize': 'romanize', 'manupulative': 'manipulative',\n                  'manjha': 'mania', 'mankrit': 'mank rit',\n                  'heteroromantic': 'hetero romantic', 'pulmanery': 'pulmonary', 'manpads': 'man pads',\n                  'supermaneuverable': 'super maneuverable', 'mandatkry': 'mandatory', 'armanents': 'armaments',\n                  'manipative': 'mancipative', 'himanity': 'humanity', 'maneuever': 'maneuver',\n                  'Kumarmangalam': 'Kumar mangalam', 'Brahmanwadi': 'Brahman wadi',\n                  'exserviceman': 'ex serviceman',\n                  'managewp': 'managed', 'manies': 'many', 'recordermans': 'recorder mans',\n                  'Feymann': 'Heymann', 'salemmango': 'salem mango', 'manufraturing': 'manufacturing',\n                  'sreeman': 'freeman', 'tamanaa': 'Tamanac', 'chlamydomanas': 'chlamydomonas',\n                  'comandant': 'commandant', 'huemanity': 'humanity', 'manaagerial': 'managerial',\n                  'lithromantics': 'lith romantics',\n                  'geramans': 'germans', 'Nagamandala': 'Naga mandala', 'humanitariarism': 'humanitarianism',\n                  'wattman': 'watt man', 'salesmanago': 'salesman ago', 'Washwoman': 'Wash woman',\n                  'rammandir': 'ram mandir', 'nomanclature': 'nomenclature', 'Haufman': 'Kaufman',\n                  'prefomance': 'performance', 'ramanunjan': 'Ramanujan', 'Freemansonry': 'Freemasonry',\n                  'supermaneuverability': 'super maneuverability', 'manstruate': 'menstruate',\n                  'Tarumanagara': 'Taruma nagara', 'RomanceTale': 'Romance Tale', 'heteromantic': 'hete romantic',\n                  'terimanals': 'terminals', 'womansplaining': 'wo mansplaining',\n                  'performancelearning': 'performance learning', 'sociomantic': 'sciomantic',\n                  'batmanvoice': 'batman voice', 'PerformanceTesting': 'Performance Testing',\n                  'manorialism': 'manorial ism', 'newscommando': 'news commando',\n                  'Entwicklungsroman': 'Entwicklungs roman',\n                  'Kunstlerroman': 'Kunstler roman', 'bodhidharman': 'Bodhidharma', 'Howmaney': 'How maney',\n                  'manufucturing': 'manufacturing', 'remmaning': 'remaining', 'rangeman': 'range man',\n                  'mythomaniac': 'mythomania', 'katgmandu': 'katmandu',\n                  'Superowoman': 'Superwoman', 'Rahmanland': 'Rahman land', 'Dormmanu': 'Dormant',\n                  'Geftman': 'Gentman', 'manufacturig': 'manufacturing', 'bramanistic': 'Brahmanistic',\n                  'padmanabhanagar': 'padmanabhan agar', 'homoromantic': 'homo romantic', 'femanists': 'feminists',\n                  'demihuman': 'demi human', 'manrega': 'Manresa', 'Pasmanda': 'Pas manda',\n                  'manufacctured': 'manufactured', 'remaninder': 'remainder', 'Marimanga': 'Mari manga',\n                  'Sloatman': 'Sloat man', 'manlet': 'man let', 'perfoemance': 'performance',\n                  'mangolian': 'mongolian', 'mangekyu': 'mange kyu', 'mansatory': 'mandatory',\n                  'managemebt': 'management', 'manufctures': 'manufactures', 'Bramanical': 'Brahmanical',\n                  'manaufacturing': 'manufacturing', 'Lakhsman': 'Lakhs man', 'Sarumans': 'Sarum ans',\n                  'mangalasutra': 'mangalsutra', 'Germanised': 'German ised',\n                  'managersworking': 'managers working', 'cammando': 'commando', 'mandrillaris': 'mandrill aris',\n                  'Emmanvel': 'Emmarvel', 'manupalation': 'manipulation', 'welcomeromanian': 'welcome romanian',\n                  'humanfemale': 'human female', 'mankirt': 'mankind', 'Haffmann': 'Hoffmann',\n                  'Panromantic': 'Pan romantic', 'demantion': 'detention', 'Suparwoman': 'Superwoman',\n                  'parasuramans': 'parasuram ans', 'sulmann': 'Suilmann', 'Shubman': 'Subman',\n                  'manspread': 'man spread', 'mandingan': 'Mandingan', 'mandalikalu': 'mandalika lu',\n                  'manufraturer': 'manufacturer', 'Wedgieman': 'Wedgie man', 'manwues': 'manages',\n                  'humanzees': 'human zees', 'Steymann': 'Stedmann', 'Jobberman': 'Jobber man',\n                  'maniquins': 'mani quins', 'biromantical': 'bi romantical', 'Rovman': 'Roman',\n                  'pyromantic': 'pyro mantic', 'Tastaman': 'Rastaman', 'Spoolman': 'Spool man',\n                  'Subramaniyan': 'Subramani yan', 'abhimana': 'abhiman a', 'manholding': 'man holding',\n                  'seviceman': 'serviceman', 'womansplained': 'womans plained', 'manniya': 'mania',\n                  'Bhraman': 'Braman', 'Laakman': 'Layman', 'mansturbate': 'masturbate',\n                  'Sulamaniya': 'Sulamani ya', 'demanters': 'decanters', 'postmanare': 'postman are',\n                  'mannualy': 'annual', 'rstman': 'Rotman', 'permanentjobs': 'permanent jobs',\n                  'Allmang': 'All mang', 'TradeCommander': 'Trade Commander', 'BasedStickman': 'Based Stickman',\n                  'Deshabhimani': 'Desha bhimani', 'manslamming': 'mans lamming', 'Brahmanwad': 'Brahman wad',\n                  'fundemantally': 'fundamentally', 'supplemantary': 'supplementary', 'egomanias': 'ego manias',\n                  'manvantar': 'Manvantara', 'spymania': 'spy mania', 'mangonada': 'mango nada',\n                  'manthras': 'mantras', 'Humanpark': 'Human park', 'manhuas': 'mahuas',\n                  'manterrupting': 'interrupting', 'dermatillomaniac': 'dermatillomania',\n                  'performancies': 'performances', 'manipulant': 'manipulate',\n                  'painterman': 'painter man', 'mangalik': 'manglik',\n                  'Neurosemantics': 'Neuro semantics', 'discrimantion': 'discrimination',\n                  'Womansplaining': 'feminist', 'mongodump': 'mongo dump', 'roadgods': 'road gods',\n                  'Oligodendraglioma': 'Oligodendroglioma', 'unrightly': 'un rightly', 'Janewright': 'Jane wright',\n                  ' righten ': ' tighten ', 'brightiest': 'brightest',\n                  'frighter': 'fighter', 'righteouness': 'righteousness', 'triangleright': 'triangle right',\n                  'Brightspace': 'Brights pace', 'techinacal': 'technical', 'chinawares': 'china wares',\n                  'Vancouever': 'Vancouver', 'cheverlet': 'cheveret', 'deverstion': 'diversion',\n                  'everbodys': 'everybody', 'Dramafever': 'Drama fever', 'reverificaton': 'reverification',\n                  'canterlever': 'canter lever', 'keywordseverywhere': 'keywords everywhere',\n                  'neverunlearned': 'never unlearned', 'everyfirst': 'every first',\n                  'neverhteless': 'nevertheless', 'clevercoyote': 'clever coyote', 'irrevershible': 'irreversible',\n                  'achievership': 'achievers hip', 'easedeverything': 'eased everything', 'youbever': 'you bever',\n                  'everperson': 'ever person', 'everydsy': 'everyday', 'whemever': 'whenever',\n                  'everyonr': 'everyone', 'severiity': 'severity', 'narracist': 'nar racist',\n                  'racistly': 'racist', 'takesuch': 'take such', 'mystakenly': 'mistakenly',\n                  'shouldntake': 'shouldnt take', 'Kalitake': 'Kali take', 'msitake': 'mistake',\n                  'straitstimes': 'straits times', 'timefram': 'timeframe', 'watchtime': 'watch time',\n                  'timetraveling': 'timet raveling', 'peactime': 'peacetime', 'timetabe': 'timetable',\n                  'cooktime': 'cook time', 'blocktime': 'block time', 'timesjobs': 'times jobs',\n                  'timesence': 'times ence', 'Touchtime': 'Touch time', 'timeloop': 'time loop',\n                  'subcentimeter': 'sub centimeter', 'timejobs': 'time jobs', 'Guardtime': 'Guard time',\n                  'realtimepolitics': 'realtime politics', 'loadingtimes': 'loading times',\n                  'timesnow': '24-hour English news channel in India', 'timesspark': 'times spark',\n                  'timetravelling': 'timet ravelling',\n                  'antimeter': 'anti meter', 'timewaste': 'time waste', 'cryptochristians': 'crypto christians',\n                  'Whatcould': 'What could', 'becomesdouble': 'becomes double', 'deathbecomes': 'death becomes',\n                  'youbecome': 'you become', 'greenseer': 'people who possess the magical ability',\n                  'rseearch': 'research', 'homeseek': 'home seek',\n                  'Greenseer': 'people who possess the magical ability', 'starseeders': 'star seeders',\n                  'seekingmillionaire': 'seeking millionaire', 'see\\u202c': 'see',\n                  'seeies': 'series', 'CodeAgon': 'Code Agon',\n                  'royago': 'royal', 'Dragonkeeper': 'Dragon keeper', 'mcgreggor': 'McGregor',\n                  'catrgory': 'category', 'Dragonknight': 'Dragon knight', 'Antergos': 'Anteros',\n                  'togofogo': 'togo fogo', 'mongorestore': 'mongo restore', 'gorgops': 'gorgons',\n                  'withgoogle': 'with google', 'goundar': 'Gondar', 'algorthmic': 'algorithmic',\n                  'goatnuts': 'goat nuts', 'vitilgo': 'vitiligo', 'polygony': 'poly gony',\n                  'digonals': 'diagonals', 'Luxemgourg': 'Luxembourg', 'UCSanDiego': 'UC SanDiego',\n                  'Ringostat': 'Ringo stat', 'takingoff': 'taking off', 'MongoImport': 'Mongo Import',\n                  'alggorithms': 'algorithms', 'dragonknight': 'dragon knight', 'negotiatior': 'negotiation',\n                  'gomovies': 'go movies', 'Withgott': 'Without',\n                  'categoried': 'categories', 'Stocklogos': 'Stock logos', 'Pedogogical': 'Pedological',\n                  'Wedugo': 'Wedge', 'golddig': 'gold dig', 'goldengroup': 'golden group',\n                  'merrigo': 'merligo', 'googlemapsAPI': 'googlemaps API', 'goldmedal': 'gold medal',\n                  'golemized': 'polemized', 'Caligornia': 'California', 'unergonomic': 'un ergonomic',\n                  'fAegon': 'wagon', 'vertigos': 'vertigo s', 'trigonomatry': 'trigonometry',\n                  'hypogonadic': 'hypogonadia', 'Mogolia': 'Mongolia', 'governmaent': 'government',\n                  'ergotherapy': 'ergo therapy', 'Bogosort': 'Bogo sort', 'goalwise': 'goal wise',\n                  'alogorithms': 'algorithms', 'MercadoPago': 'Mercado Pago', 'rivigo': 'rivi go',\n                  'govshutdown': 'gov shutdown', 'gorlfriend': 'girlfriend',\n                  'stategovt': 'state govt', 'Chickengonia': 'Chicken gonia', 'Yegorovich': 'Yegorov ich',\n                  'regognitions': 'recognitions', 'gorichen': 'Gori Chen Mountain',\n                  'goegraphies': 'geographies', 'gothras': 'goth ras', 'belagola': 'bela gola',\n                  'snapragon': 'snapdragon', 'oogonial': 'oogonia l', 'Amigofoods': 'Amigo foods',\n                  'Sigorn': 'son of Styr', 'algorithimic': 'algorithmic',\n                  'innermongolians': 'inner mongolians', 'ArangoDB': 'Arango DB', 'zigolo': 'gigolo',\n                  'regognized': 'recognized', 'Moongot': 'Moong ot', 'goldquest': 'gold quest',\n                  'catagorey': 'category', 'got7': 'got', 'jetbingo': 'jet bingo', 'Dragonchain': 'Dragon chain',\n                  'catwgorized': 'categorized', 'gogoro': 'gogo ro', 'Tobagoans': 'Tobago ans',\n                  'digonal': 'di gonal', 'algoritmic': 'algorismic', 'dragonflag': 'dragon flag',\n                  'Indigoflight': 'Indigo flight',\n                  'governening': 'governing', 'ergosphere': 'ergo sphere',\n                  'pingo5': 'pingo', 'Montogo': 'montego', 'Rivigo': 'technology-enabled logistics company',\n                  'Jigolo': 'Gigolo', 'phythagoras': 'pythagoras', 'Mangolian': 'Mongolian',\n                  'forgottenfaster': 'forgotten faster', 'stargold': 'a Hindi movie channel',\n                  'googolplexain': 'googolplexian', 'corpgov': 'corp gov',\n                  'govtribe': 'provides real-time federal contracting market intel',\n                  'dragonglass': 'dragon glass', 'gorakpur': 'Gorakhpur', 'MangoPay': 'Mango Pay',\n                  'chigoe': 'sub-tropical climates', 'BingoBox': 'an investment company', '走go': 'go',\n                  'followingorder': 'following order', 'pangolinminer': 'pangolin miner',\n                  'negosiation': 'negotiation', 'lexigographers': 'lexicographers', 'algorithom': 'algorithm',\n                  'unforgottable': 'unforgettable', 'wellsfargoemail': 'wellsfargo email',\n                  'daigonal': 'diagonal', 'Pangoro': 'cantankerous Pokemon', 'negotiotions': 'negotiations',\n                  'Swissgolden': 'Swiss golden', 'google4': 'google', 'Agoraki': 'Ago raki',\n                  'Garthago': 'Carthago', 'Stegosauri': 'stegosaurus', 'ergophobia': 'ergo phobia',\n                  'bigolive': 'big olive', 'bittergoat': 'bitter goat', 'naggots': 'faggots',\n                  'googology': 'online encyclopedia', 'algortihms': 'algorithms', 'bengolis': 'Bengalis',\n                  'fingols': 'Finnish people are supposedly descended from Mongols',\n                  'savethechildren': 'save thechildren',\n                  'stopings': 'stoping', 'stopsits': 'stop sits', 'stopsigns': 'stop signs',\n                  'Galastop': 'Galas top', 'pokestops': 'pokes tops', 'forcestop': 'forces top',\n                  'Hopstop': 'Hops top', 'stoppingexercises': 'stopping exercises', 'coinstop': 'coins top',\n                  'stoppef': 'stopped', 'workaway': 'work away', 'snazzyway': 'snazzy way',\n                  'Rewardingways': 'Rewarding ways', 'cloudways': 'cloud ways', 'Cloudways': 'Cloud ways',\n                  'Brainsway': 'Brains way', 'nesraway': 'nearaway',\n                  'AlwaysHired': 'Always Hired', 'expessway': 'expressway', 'Syncway': 'Sync way',\n                  'LeewayHertz': 'Blockchain Company', 'towayrds': 'towards', 'swayable': 'sway able',\n                  'Telloway': 'Tello way', 'palsmodium': 'plasmodium', 'Gobackmodi': 'Goback modi',\n                  'comodies': 'corodies', 'islamphobic': 'islam phobic', 'islamphobia': 'islam phobia',\n                  'citiesbetter': 'cities better', 'betterv3': 'better', 'betterDtu': 'better Dtu',\n                  'Babadook': 'a horror drama film', 'Ahemadabad': 'Ahmadabad', 'faidabad': 'Faizabad',\n                  'Amedabad': 'Ahmedabad', 'kabadii': 'kabaddi', 'badmothing': 'badmouthing',\n                  'badminaton': 'badminton', 'badtameezdil': 'badtameez dil', 'badeffects': 'bad effects',\n                  '∠bad': 'bad', 'ahemadabad': 'Ahmadabad', 'embaded': 'embased', 'Isdhanbad': 'Is dhanbad',\n                  'badgermoles': 'enormous, blind mammal', 'allhabad': 'Allahabad', 'ghazibad': 'ghazi bad',\n                  'htderabad': 'Hyderabad', 'Auragabad': 'Aurangabad', 'ahmedbad': 'Ahmedabad',\n                  'ahmdabad': 'Ahmadabad', 'alahabad': 'Allahabad',\n                  'Hydeabad': 'Hyderabad', 'Gyroglove': 'wearable technology', 'foodlovee': 'food lovee',\n                  'slovenised': 'slovenia', 'handgloves': 'hand gloves', 'lovestep': 'love step',\n                  'lovejihad': 'love jihad', 'RolloverBox': 'Rollover Box', 'stupidedt': 'stupidest',\n                  'toostupid': 'too stupid',\n                  'pakistanisbeautiful': 'pakistanis beautiful', 'ispakistan': 'is pakistan',\n                  'inpersonations': 'impersonations', 'medicalperson': 'medical person',\n                  'interpersonation': 'inter personation', 'workperson': 'work person',\n                  'personlich': 'person lich', 'persoenlich': 'person lich',\n                  'middleperson': 'middle person', 'personslized': 'personalized',\n                  'personifaction': 'personification', 'welcomemarriage': 'welcome marriage',\n                  'come2': 'come to', 'upcomedians': 'up comedians', 'overvcome': 'overcome',\n                  'talecome': 'tale come', 'cometitive': 'competitive', 'arencome': 'aren come',\n                  'achecomes': 'ache comes', '」come': 'come',\n                  'comepleted': 'completed', 'overcomeanxieties': 'overcome anxieties',\n                  'demigirl': 'demi girl', 'gridgirl': 'female models of the race', 'halfgirlfriend': 'half girlfriend',\n                  'girlriend': 'girlfriend', 'fitgirl': 'fit girl', 'girlfrnd': 'girlfriend', 'awrong': 'aw rong',\n                  'northcap': 'north cap', 'productionsupport': 'production support',\n                  'Designbold': 'Online Photo Editor Design Studio',\n                  'skyhold': 'sky hold', 'shuoldnt': 'shouldnt', 'anarold': 'Android', 'yaerold': 'year old',\n                  'soldiders': 'soldiers', 'indrold': 'Android', 'blindfoldedly': 'blindfolded',\n                  'overcold': 'over cold', 'Goldmont': 'microarchitecture in Intel', 'boldspot': 'bolds pot',\n                  'Rankholders': 'Rank holders', 'cooldrink': 'cool drink', 'beltholders': 'belt holders',\n                  'GoldenDict': 'open-source dictionary program', 'softskill': 'softs kill',\n                  'Cooldige': 'the 30th president of the United States',\n                  'newkiller': 'new killer', 'skillselect': 'skills elect', 'nonskilled': 'non skilled',\n                  'killyou': 'kill you', 'Skillport': 'Army e-Learning Program', 'unkilled': 'un killed',\n                  'killikng': 'killing', 'killograms': 'kilograms',\n                  'Worldkillers': 'World killers', 'reskilled': 'skilled',\n                  'killedshivaji': 'killed shivaji', 'honorkillings': 'honor killings',\n                  'skillclasses': 'skill classes', 'microskills': 'micros kills',\n                  'Skillselect': 'Skills elect', 'ratkill': 'rat kill',\n                  'pleasegive': 'please give', 'flashgive': 'flash give',\n                  'southerntelescope': 'southern telescope', 'westsouth': 'west south',\n                  'southAfricans': 'south Africans', 'Joboutlooks': 'Job outlooks', 'joboutlook': 'job outlook',\n                  'Outlook365': 'Outlook 365', 'Neulife': 'Neu life', 'qualifeid': 'qualified',\n                  'nullifed': 'nullified', 'lifeaffect': 'life affect', 'lifestly': 'lifestyle',\n                  'aristocracylifestyle': 'aristocracy lifestyle', 'antilife': 'anti life',\n                  'afterafterlife': 'after afterlife', 'lifestylye': 'lifestyle', 'prelife': 'pre life',\n                  'lifeute': 'life ute', 'liferature': 'literature',\n                  'securedlife': 'secured life', 'doublelife': 'double life', 'antireligion': 'anti religion',\n                  'coreligionist': 'co religionist', 'petrostates': 'petro states', 'otherstates': 'others tates',\n                  'spacewithout': 'space without', 'withoutyou': 'without you',\n                  'withoutregistered': 'without registered', 'weightwithout': 'weight without',\n                  'withoutcheck': 'without check', 'milkwithout': 'milk without',\n                  'Highschoold': 'High school', 'memoney': 'money', 'moneyof': 'mony of', 'Oneplus': 'OnePlus',\n                  'OnePlus': 'Chinese smartphone manufacturer', 'Beerus': 'the God of Destruction',\n                  'takeoverr': 'takeover', 'demonetizedd': 'demonetized', 'polyhouse': 'Polytunnel',\n                  'Elitmus': 'eLitmus', 'eLitmus': 'Indian company that helps companies in hiring employees',\n                  'becone': 'become', 'nestaway': 'nest away', 'takeoverrs': 'takeovers', 'Istop': 'I stop',\n                  'Austira': 'Australia', 'germeny': 'Germany', 'mansoon': 'man soon',\n                  'worldmax': 'wholesaler of drum parts',\n                  'ammusement': 'amusement', 'manyare': 'many are', 'supplymentary': 'supply mentary',\n                  'timesup': 'times up', 'homologus': 'homologous', 'uimovement': 'ui movement', 'spause': 'spouse',\n                  'aesexual': 'asexual', 'Iovercome': 'I overcome', 'developmeny': 'development',\n                  'hindusm': 'hinduism', 'sexpat': 'sex tourism', 'sunstop': 'sun stop', 'polyhouses': 'Polytunnel',\n                  'usefl': 'useful', 'Fundamantal': 'fundamental', 'environmentai': 'environmental',\n                  'Redmi': 'Xiaomi Mobile', 'Loy Machedo': ' Motivational Speaker ', 'unacademy': 'Unacademy',\n                  'Boruto': 'Naruto Next Generations', 'Upwork': 'Up work',\n                  'Unacademy': 'educational technology company',\n                  'HackerRank': 'Hacker Rank', 'upwork': 'up work', 'Chromecast': 'Chrome cast',\n                  'microservices': 'micro services', 'Undertale': 'video game', 'undergraduation': 'under graduation',\n                  'chapterwise': 'chapter wise', 'twinflame': 'twin flame', 'Hotstar': 'Hot star',\n                  'blockchains': 'blockchain',\n                  'darkweb': 'dark web', 'Microservices': 'Micro services', 'Nearbuy': 'Nearby',\n                  ' Padmaavat ': ' Padmavati ', ' padmavat ': ' Padmavati ', ' Padmaavati ': ' Padmavati ',\n                  ' Padmavat ': ' Padmavati ', ' internshala ': ' internship and online training platform in India ',\n                  'dream11': ' fantasy sports platform in India ', 'conciousnesss': 'consciousnesses',\n                  'Dream11': ' fantasy sports platform in India ', 'cointry': 'country', ' coinvest ': ' invest ',\n                  '23 andme': 'privately held personal genomics and biotechnology company in California',\n                  'Trumpism': 'philosophy and politics espoused by Donald Trump',\n                  'Trumpian': 'viewpoints of President Donald Trump', 'Trumpists': 'admirer of Donald Trump',\n                  'coincidents': 'coincidence', 'coinsized': 'coin sized', 'coincedences': 'coincidences',\n                  'cointries': 'countries', 'coinsidered': 'considered', 'coinfirm': 'confirm',\n                  'humilates':'humiliates', 'vicevice':'vice vice', 'politicak':'political', 'Sumaterans':'Sumatrans',\n                  'Kamikazis':'Kamikazes', 'unmoraled':'unmoral', 'eduacated':'educated', 'moraled':'morale',\n                  'Amharc':'Amarc', 'where Burkhas':'wear Burqas', 'Baloochistan':'Balochistan', 'durgahs':'durgans',\n                  'illigitmate':'illegitimate', 'hillum':'helium','treatens':'threatens','mutiliating':'mutilating',\n                  'speakingly':'speaking', 'pretex':'pretext', 'menstruateion':'menstruation', \n                  'genocidizing':'genociding', 'maratis':'Maratism','Parkistinian':'Pakistani', 'SPEICIAL':'SPECIAL',\n                  'REFERNECE':'REFERENCE', 'provocates':'provokes', 'FAMINAZIS':'FEMINAZIS', 'repugicans':'republicans',\n                  'tonogenesis':'tone', 'winor':'win', 'redicules':'ridiculous', 'Beluchistan':'Balochistan', \n                  'volime':'volume', 'namaj':'namaz', 'CONgressi':'Congress', 'Ashifa':'Asifa', 'queffing':'queefing',\n                  'montheistic':'nontheistic', 'Rajsthan':'Rajasthan', 'Rajsthanis':'Rajasthanis', 'specrum':'spectrum',\n                  'brophytes':'bryophytes', 'adhaar':'Adhara', 'slogun':'slogan', 'harassd':'harassed',\n                  'transness':'trans gender', 'Insdians':'Indians', 'Trampaphobia':'Trump aphobia', 'attrected':'attracted',\n                  'Yahtzees':'Yahtzee', 'thiests':'atheists', 'thrir':'their', 'extraterestrial':'extraterrestrial',\n                  'silghtest':'slightest', 'primarty':'primary','brlieve':'believe', 'fondels':'fondles',\n                  'loundly':'loudly', 'bootythongs':'booty thongs', 'understamding':'understanding', 'degenarate':'degenerate',\n                  'narsistic':'narcistic', 'innerskin':'inner skin','spectulated':'speculated', 'hippocratical':'Hippocratical',\n                  'itstead':'instead', 'parralels':'parallels', 'sloppers':'slippers'\n                  }\n\ndef clean_bad_case_words(text):\n    for bad_word in bad_case_words:\n        if bad_word in text:\n            text = text.replace(bad_word, bad_case_words[bad_word])\n    return text","metadata":{"execution":{"iopub.status.busy":"2022-10-26T03:15:03.544226Z","iopub.execute_input":"2022-10-26T03:15:03.544503Z","iopub.status.idle":"2022-10-26T03:15:03.706317Z","shell.execute_reply.started":"2022-10-26T03:15:03.544479Z","shell.execute_reply":"2022-10-26T03:15:03.705241Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mis_connect_list = ['(W|w)hat', '(W|w)hy', '(H|h)ow', '(W|w)hich', '(W|w)here', '(W|w)ill']\nmis_connect_re = re.compile('(%s)' % '|'.join(mis_connect_list))\n\nmis_spell_mapping = {'whattsup': 'WhatsApp', 'whatasapp':'WhatsApp', 'whatsupp':'WhatsApp', \n                      'whatcus':'what cause', 'arewhatsapp': 'are WhatsApp', 'Hwhat':'what',\n                      'Whwhat': 'What', 'whatshapp':'WhatsApp', 'howhat':'how that',\n                      # why\n                      'Whybis':'Why is', 'laowhy86':'Foreigners who do not respect China',\n                      'Whyco-education':'Why co-education',\n                      # How\n                      \"Howddo\":\"How do\", 'Howeber':'However', 'Showh':'Show',\n                      \"Willowmagic\":'Willow magic', 'WillsEye':'Will Eye', 'Williby':'will by'}\ndef spacing_some_connect_words(text):\n    \"\"\"\n    'Whyare' -> 'Why are'\n    \"\"\"\n    ori = text\n    for error in mis_spell_mapping:\n        if error in text:\n            text = text.replace(error, mis_spell_mapping[error])\n            \n    # what\n    text = re.sub(r\" (W|w)hat+(s)*[A|a]*(p)+ \", \" WhatsApp \", text)\n    text = re.sub(r\" (W|w)hat\\S \", \" What \", text)\n    text = re.sub(r\" \\S(W|w)hat \", \" What \", text)\n    # why\n    text = re.sub(r\" (W|w)hy\\S \", \" Why \", text)\n    text = re.sub(r\" \\S(W|w)hy \", \" Why \", text)\n    # How\n    text = re.sub(r\" (H|h)ow\\S \", \" How \", text)\n    text = re.sub(r\" \\S(H|h)ow \", \" How \", text)\n    # which\n    text = re.sub(r\" (W|w)hich\\S \", \" Which \", text)\n    text = re.sub(r\" \\S(W|w)hich \", \" Which \", text)\n    # where\n    text = re.sub(r\" (W|w)here\\S \", \" Where \", text)\n    text = re.sub(r\" \\S(W|w)here \", \" Where \", text)\n    # \n    text = mis_connect_re.sub(r\" \\1 \", text)\n    text = text.replace(\"What sApp\", 'WhatsApp')\n    \n    text = remove_space(text)\n    return text","metadata":{"execution":{"iopub.status.busy":"2022-10-26T03:15:03.708071Z","iopub.execute_input":"2022-10-26T03:15:03.708788Z","iopub.status.idle":"2022-10-26T03:15:03.719725Z","shell.execute_reply.started":"2022-10-26T03:15:03.708749Z","shell.execute_reply":"2022-10-26T03:15:03.718716Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def clean_repeat_words(text):\n    text = text.replace(\"img\", \"ing\")\n\n    text = re.sub(r\"(I|i)(I|i)+ng\", \"ing\", text)\n    text = re.sub(r\"(L|l)(L|l)(L|l)+y\", \"lly\", text)\n    text = re.sub(r\"(A|a)(A|a)(A|a)+\", \"a\", text)\n    text = re.sub(r\"(C|c)(C|c)(C|c)+\", \"cc\", text)\n    text = re.sub(r\"(D|d)(D|d)(D|d)+\", \"dd\", text)\n    text = re.sub(r\"(E|e)(E|e)(E|e)+\", \"ee\", text)\n    text = re.sub(r\"(F|f)(F|f)(F|f)+\", \"ff\", text)\n    text = re.sub(r\"(G|g)(G|g)(G|g)+\", \"gg\", text)\n    text = re.sub(r\"(I|i)(I|i)(I|i)+\", \"i\", text)\n    text = re.sub(r\"(K|k)(K|k)(K|k)+\", \"k\", text)\n    text = re.sub(r\"(L|l)(L|l)(L|l)+\", \"ll\", text)\n    text = re.sub(r\"(M|m)(M|m)(M|m)+\", \"mm\", text)\n    text = re.sub(r\"(N|n)(N|n)(N|n)+\", \"nn\", text)\n    text = re.sub(r\"(O|o)(O|o)(O|o)+\", \"oo\", text)\n    text = re.sub(r\"(P|p)(P|p)(P|p)+\", \"pp\", text)\n    text = re.sub(r\"(Q|q)(Q|q)+\", \"q\", text)\n    text = re.sub(r\"(R|r)(R|r)(R|r)+\", \"rr\", text)\n    text = re.sub(r\"(S|s)(S|s)(S|s)+\", \"ss\", text)\n    text = re.sub(r\"(T|t)(T|t)(T|t)+\", \"tt\", text)\n    text = re.sub(r\"(V|v)(V|v)+\", \"v\", text)\n    text = re.sub(r\"(Y|y)(Y|y)(Y|y)+\", \"y\", text)\n    text = re.sub(r\"plzz+\", \"please\", text)\n    text = re.sub(r\"(Z|z)(Z|z)(Z|z)+\", \"zz\", text)\n    return text","metadata":{"execution":{"iopub.status.busy":"2022-10-26T03:15:03.721318Z","iopub.execute_input":"2022-10-26T03:15:03.721752Z","iopub.status.idle":"2022-10-26T03:15:03.734275Z","shell.execute_reply.started":"2022-10-26T03:15:03.721716Z","shell.execute_reply":"2022-10-26T03:15:03.733350Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import psutil\nfrom multiprocessing import Pool\n\nnum_partitions = 20  # number of partitions to split dataframe\nnum_cores = psutil.cpu_count()  # number of cores on your machine\n\nprint('number of cores:', num_cores)\ndef df_parallelize_run(df, func):\n    df_split = np.array_split(df, num_partitions)\n    pool = Pool(num_cores)\n    df = pd.concat(pool.map(func, df_split))\n    pool.close()\n    pool.join()\n    return df","metadata":{"execution":{"iopub.status.busy":"2022-10-26T03:15:03.735649Z","iopub.execute_input":"2022-10-26T03:15:03.736116Z","iopub.status.idle":"2022-10-26T03:15:03.748041Z","shell.execute_reply.started":"2022-10-26T03:15:03.736078Z","shell.execute_reply":"2022-10-26T03:15:03.746843Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def preprocess(text):\n    \"\"\"\n    preprocess text main steps\n    \"\"\"\n    text = remove_space(text)\n    text = clean_special_punctuations(text)\n    text = clean_number(text)\n    text = pre_clean_bad_words(text)\n#     text = decontracted(text)\n    text = clean_latex(text)\n    text = clean_misspell(text)\n    text = spacing_punctuation(text)\n    text = spacing_some_connect_words(text)\n    text = clean_bad_case_words(text)\n    text = clean_repeat_words(text)\n    text = remove_space(text)\n    return text\n\ndef text_clean_wrapper(df):\n    df[\"question_text\"] = df[\"question_text\"].apply(preprocess)\n    return df","metadata":{"execution":{"iopub.status.busy":"2022-10-26T03:15:03.749849Z","iopub.execute_input":"2022-10-26T03:15:03.750275Z","iopub.status.idle":"2022-10-26T03:15:03.759145Z","shell.execute_reply.started":"2022-10-26T03:15:03.750239Z","shell.execute_reply":"2022-10-26T03:15:03.758157Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data = df_parallelize_run(train_data, text_clean_wrapper)","metadata":{"execution":{"iopub.status.busy":"2022-10-26T03:15:03.760809Z","iopub.execute_input":"2022-10-26T03:15:03.761174Z","iopub.status.idle":"2022-10-26T03:22:45.980356Z","shell.execute_reply.started":"2022-10-26T03:15:03.761141Z","shell.execute_reply":"2022-10-26T03:22:45.978987Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.head()","metadata":{"execution":{"iopub.status.busy":"2022-10-26T03:22:45.982636Z","iopub.execute_input":"2022-10-26T03:22:45.983065Z","iopub.status.idle":"2022-10-26T03:22:45.996110Z","shell.execute_reply.started":"2022-10-26T03:22:45.983020Z","shell.execute_reply":"2022-10-26T03:22:45.995122Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cur_vocabulary = set()\nfor text in tqdm(train_data['question_text'].values.tolist()):\n    words = text.split(' ')\n    cur_vocabulary.update(set(words))\n\nbug_punc_spacing_words_mapping = {}\nfor vocab in cur_vocabulary:\n    if '-' in vocab:\n        # whether the glove or para contain this word\n        if (vocab in embed_glove or vocab.capitalize() in embed_glove or vocab.lower() in embed_glove):\n            bug_punc_spacing_words_mapping[f\" {' - '.join(vocab.split('-'))} \"] = f\" {vocab} \"\n    \n    if '.' in vocab:\n        if vocab.endswith('.'):\n            continue\n        \n        if (vocab in embed_glove or vocab.capitalize() in embed_glove or vocab.lower() in embed_glove):\n            bug_punc_spacing_words_mapping[f\" {' . '.join(vocab.split('.'))} \"] = f\" {vocab} \"\n                                    \n# del bug_punc_spacing_words_mapping['  -  ']\nprint(f'found {len(bug_punc_spacing_words_mapping)} bug words')","metadata":{"execution":{"iopub.status.busy":"2022-10-26T03:22:45.997909Z","iopub.execute_input":"2022-10-26T03:22:45.998731Z","iopub.status.idle":"2022-10-26T03:22:51.006387Z","shell.execute_reply.started":"2022-10-26T03:22:45.998691Z","shell.execute_reply":"2022-10-26T03:22:51.005272Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def spacing_dash_point(text):\n    if '-' in text:\n        text = text.replace('-', ' - ')\n    if '.' in text:\n        text = text.replace('.', ' . ')\n    return text","metadata":{"execution":{"iopub.status.busy":"2022-10-26T03:22:51.008088Z","iopub.execute_input":"2022-10-26T03:22:51.008504Z","iopub.status.idle":"2022-10-26T03:22:51.014455Z","shell.execute_reply.started":"2022-10-26T03:22:51.008464Z","shell.execute_reply":"2022-10-26T03:22:51.013311Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data[\"question_text\"] = train_data[\"question_text\"].apply(spacing_dash_point)","metadata":{"execution":{"iopub.status.busy":"2022-10-26T03:22:51.016023Z","iopub.execute_input":"2022-10-26T03:22:51.016425Z","iopub.status.idle":"2022-10-26T03:22:51.416478Z","shell.execute_reply.started":"2022-10-26T03:22:51.016387Z","shell.execute_reply":"2022-10-26T03:22:51.415357Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def fix_dash_point_spacing_bug(text):\n    for bug_dash in bug_punc_spacing_words_mapping:\n        if bug_dash in text:\n            text = text.replace(bug_dash, bug_punc_spacing_words_mapping[bug_dash])\n    return text\n\ndef fix_dash_point_spacing_bug_wrapper(df):\n    df[\"question_text\"] = df[\"question_text\"].apply(fix_dash_point_spacing_bug)\n    return df","metadata":{"execution":{"iopub.status.busy":"2022-10-26T03:22:51.418254Z","iopub.execute_input":"2022-10-26T03:22:51.418654Z","iopub.status.idle":"2022-10-26T03:22:51.425041Z","shell.execute_reply.started":"2022-10-26T03:22:51.418613Z","shell.execute_reply":"2022-10-26T03:22:51.423996Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data = df_parallelize_run(train_data, fix_dash_point_spacing_bug_wrapper)","metadata":{"execution":{"iopub.status.busy":"2022-10-26T03:22:51.426567Z","iopub.execute_input":"2022-10-26T03:22:51.427183Z","iopub.status.idle":"2022-10-26T03:59:32.454149Z","shell.execute_reply.started":"2022-10-26T03:22:51.427143Z","shell.execute_reply":"2022-10-26T03:59:32.452738Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sentences = train_data[\"question_text\"].apply(lambda x: x.split())\nvocab = vocab_builder(sentences)","metadata":{"execution":{"iopub.status.busy":"2022-10-26T03:59:32.456008Z","iopub.execute_input":"2022-10-26T03:59:32.456441Z","iopub.status.idle":"2022-10-26T03:59:41.668986Z","shell.execute_reply.started":"2022-10-26T03:59:32.456393Z","shell.execute_reply":"2022-10-26T03:59:41.667882Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"oov = check_coverage(vocab, embed_glove)","metadata":{"execution":{"iopub.status.busy":"2022-10-26T03:59:41.670766Z","iopub.execute_input":"2022-10-26T03:59:41.671175Z","iopub.status.idle":"2022-10-26T03:59:42.376325Z","shell.execute_reply.started":"2022-10-26T03:59:41.671136Z","shell.execute_reply":"2022-10-26T03:59:42.375279Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"oov[:50]","metadata":{"execution":{"iopub.status.busy":"2022-10-26T03:59:42.378066Z","iopub.execute_input":"2022-10-26T03:59:42.378781Z","iopub.status.idle":"2022-10-26T03:59:42.388592Z","shell.execute_reply.started":"2022-10-26T03:59:42.378741Z","shell.execute_reply":"2022-10-26T03:59:42.387408Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train_data[train_data['question_text'].str.contains(\"YOUSA\")].head()","metadata":{"execution":{"iopub.status.busy":"2022-10-26T03:59:42.390242Z","iopub.execute_input":"2022-10-26T03:59:42.390735Z","iopub.status.idle":"2022-10-26T03:59:42.397890Z","shell.execute_reply.started":"2022-10-26T03:59:42.390699Z","shell.execute_reply":"2022-10-26T03:59:42.396928Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train_data.iloc[687,1]","metadata":{"execution":{"iopub.status.busy":"2022-10-26T03:59:42.399611Z","iopub.execute_input":"2022-10-26T03:59:42.399997Z","iopub.status.idle":"2022-10-26T03:59:42.407388Z","shell.execute_reply.started":"2022-10-26T03:59:42.399962Z","shell.execute_reply":"2022-10-26T03:59:42.406420Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data['question_text'] = train_data['question_text'].apply(lambda x: x.lower())","metadata":{"execution":{"iopub.status.busy":"2022-10-26T03:59:42.409020Z","iopub.execute_input":"2022-10-26T03:59:42.409572Z","iopub.status.idle":"2022-10-26T03:59:42.992380Z","shell.execute_reply.started":"2022-10-26T03:59:42.409536Z","shell.execute_reply":"2022-10-26T03:59:42.991283Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import nltk\nfrom nltk.corpus import stopwords\nimport string\n\nfrom scipy.sparse import hstack\n\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\n%matplotlib inline","metadata":{"execution":{"iopub.status.busy":"2022-10-26T03:59:42.993670Z","iopub.execute_input":"2022-10-26T03:59:42.994270Z","iopub.status.idle":"2022-10-26T03:59:43.814611Z","shell.execute_reply.started":"2022-10-26T03:59:42.994227Z","shell.execute_reply":"2022-10-26T03:59:43.813428Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data['target'].value_counts().plot(kind='bar',rot=0)","metadata":{"execution":{"iopub.status.busy":"2022-10-26T03:59:43.819863Z","iopub.execute_input":"2022-10-26T03:59:43.821487Z","iopub.status.idle":"2022-10-26T03:59:44.054588Z","shell.execute_reply.started":"2022-10-26T03:59:43.821447Z","shell.execute_reply":"2022-10-26T03:59:44.053674Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## Number of words in the text ##\ntrain_data[\"num_words\"] = train_data[\"question_text\"].apply(lambda x: len(str(x).split()))\n# test_df[\"num_words\"] = test_df[\"question_text\"].apply(lambda x: len(str(x).split()))\n\n## Number of unique words in the text ##\ntrain_data[\"num_unique_words\"] = train_data[\"question_text\"].apply(lambda x: len(set(str(x).split())))\n# test_df[\"num_unique_words\"] = test_df[\"question_text\"].apply(lambda x: len(set(str(x).split())))\n\n## Number of characters in the text ##\ntrain_data[\"num_chars\"] = train_data[\"question_text\"].apply(lambda x: len(str(x)))\n# test_df[\"num_chars\"] = test_df[\"question_text\"].apply(lambda x: len(str(x)))\n\n## Number of title case words in the text ##\ntrain_data[\"num_words_title\"] = train_data[\"question_text\"].apply(lambda x: len([w for w in str(x).split() if w.istitle()]))\n# test_df[\"num_words_title\"] = test_df[\"question_text\"].apply(lambda x: len([w for w in str(x).split() if w.istitle()]))\n\n## Average length of the words in the text ##\ntrain_data[\"mean_word_len\"] = train_data[\"question_text\"].apply(lambda x: np.mean([len(w) for w in str(x).split()]))\n# test_df[\"mean_word_len\"] = test_df[\"question_text\"].apply(lambda x: np.mean([len(w) for w in str(x).split()]))","metadata":{"execution":{"iopub.status.busy":"2022-10-26T03:59:44.056228Z","iopub.execute_input":"2022-10-26T03:59:44.056865Z","iopub.status.idle":"2022-10-26T04:00:09.407095Z","shell.execute_reply.started":"2022-10-26T03:59:44.056826Z","shell.execute_reply":"2022-10-26T04:00:09.405956Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"f, axes = plt.subplots(3, 1, figsize=(10,20))\nsns.boxplot(x='target', y='num_words', data=train_data, ax=axes[0])\naxes[0].set_xlabel('Target', fontsize=12)\naxes[0].set_title(\"Number of words in each class\", fontsize=15)\n\nsns.boxplot(x='target', y='num_unique_words', data=train_data, ax=axes[1])\naxes[1].set_xlabel('Target', fontsize=12)\naxes[1].set_title(\"Number of unique words in each class\", fontsize=15)\n\nsns.boxplot(x='target', y='num_chars', data=train_data, ax=axes[2])\naxes[2].set_xlabel('Target', fontsize=12)\naxes[2].set_title(\"Number of characters in each class\", fontsize=15)","metadata":{"execution":{"iopub.status.busy":"2022-10-26T04:00:09.408942Z","iopub.execute_input":"2022-10-26T04:00:09.409356Z","iopub.status.idle":"2022-10-26T04:00:10.865317Z","shell.execute_reply.started":"2022-10-26T04:00:09.409315Z","shell.execute_reply":"2022-10-26T04:00:10.864264Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data['num_words'].loc[train_data['num_words']>60] = 60\ntrain_data['num_unique_words'].loc[train_data['num_unique_words']>10] = 10 \ntrain_data['num_chars'].loc[train_data['num_chars']>350] = 350 \n\nf, axes = plt.subplots(3, 1, figsize=(10,20))\nsns.boxplot(x='target', y='num_words', data=train_data, ax=axes[0], showmeans=True)\naxes[0].set_xlabel('Target', fontsize=12)\naxes[0].set_title(\"Number of words in each class\", fontsize=15)\n\nsns.boxplot(x='target', y='num_unique_words', data=train_data, ax=axes[1], showmeans=True)\naxes[1].set_xlabel('Target', fontsize=12)\naxes[1].set_title(\"Number of unique words in each class\", fontsize=15)\n\nsns.boxplot(x='target', y='num_chars', data=train_data, ax=axes[2], showmeans=True)\naxes[2].set_xlabel('Target', fontsize=12)\naxes[2].set_title(\"Number of characters in each class\", fontsize=15)","metadata":{"execution":{"iopub.status.busy":"2022-10-26T04:00:10.866897Z","iopub.execute_input":"2022-10-26T04:00:10.867535Z","iopub.status.idle":"2022-10-26T04:00:12.572426Z","shell.execute_reply.started":"2022-10-26T04:00:10.867494Z","shell.execute_reply":"2022-10-26T04:00:12.571270Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"stops = set(stopwords.words(\"english\"))\ntrain_data[\"num_stopwords\"] = train_data[\"question_text\"].apply(lambda x: len([w for w in str(x).lower().split() if w in stops]))\nsns.boxplot(x='target', y='num_stopwords', data=train_data, showmeans=True)","metadata":{"execution":{"iopub.status.busy":"2022-10-26T04:00:12.573762Z","iopub.execute_input":"2022-10-26T04:00:12.574139Z","iopub.status.idle":"2022-10-26T04:00:17.688425Z","shell.execute_reply.started":"2022-10-26T04:00:12.574100Z","shell.execute_reply":"2022-10-26T04:00:17.687492Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from nltk.corpus import stopwords\ndef plot_top_stopwords(text):\n    stop = set(stopwords.words('english'))\n    new= text.str.split()\n    new=new.values.tolist()\n    corpus=[word for i in new for word in i]\n    from collections import defaultdict\n    dic=defaultdict(int)\n    for word in corpus:\n        if word in stop:\n            dic[word]+=1\n            \n    top=sorted(dic.items(), key=lambda x:x[1],reverse=True)[:10] \n    x,y=zip(*top)\n    plt.bar(x,y)","metadata":{"execution":{"iopub.status.busy":"2022-10-26T04:00:17.689777Z","iopub.execute_input":"2022-10-26T04:00:17.690269Z","iopub.status.idle":"2022-10-26T04:00:17.698445Z","shell.execute_reply.started":"2022-10-26T04:00:17.690228Z","shell.execute_reply":"2022-10-26T04:00:17.697482Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_top_stopwords(train_data['question_text'])","metadata":{"execution":{"iopub.status.busy":"2022-10-26T04:00:17.700014Z","iopub.execute_input":"2022-10-26T04:00:17.700711Z","iopub.status.idle":"2022-10-26T04:00:26.938243Z","shell.execute_reply.started":"2022-10-26T04:00:17.700668Z","shell.execute_reply":"2022-10-26T04:00:26.937252Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from collections import Counter\ndef plot_top_n_non_stopwords(text):\n    stop = set(stopwords.words('english'))\n    new= text.str.split()\n    new=new.values.tolist()\n    corpus=[word for i in new for word in i]\n\n    counter=Counter(corpus)\n    most=counter.most_common()\n    x, y=[], []\n    for word,count in most[:10]:\n        if (word not in stop):\n            x.append(word)\n            y.append(count)\n            \n    sns.barplot(x=y,y=x)","metadata":{"execution":{"iopub.status.busy":"2022-10-26T04:00:26.939621Z","iopub.execute_input":"2022-10-26T04:00:26.940244Z","iopub.status.idle":"2022-10-26T04:00:26.950094Z","shell.execute_reply.started":"2022-10-26T04:00:26.940183Z","shell.execute_reply":"2022-10-26T04:00:26.949141Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def tokenization(text):\n    tokens = nltk.word_tokenize(text)\n    return [w for w in tokens if w.isalpha()]","metadata":{"execution":{"iopub.status.busy":"2022-10-26T04:00:26.951600Z","iopub.execute_input":"2022-10-26T04:00:26.952091Z","iopub.status.idle":"2022-10-26T04:00:26.960519Z","shell.execute_reply.started":"2022-10-26T04:00:26.952056Z","shell.execute_reply":"2022-10-26T04:00:26.959477Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data['question_text'] = train_data['question_text'].apply(lambda x: tokenization(x))\ntrain_data.head()","metadata":{"execution":{"iopub.status.busy":"2022-10-26T04:00:26.962613Z","iopub.execute_input":"2022-10-26T04:00:26.962899Z","iopub.status.idle":"2022-10-26T04:03:15.109045Z","shell.execute_reply.started":"2022-10-26T04:00:26.962874Z","shell.execute_reply":"2022-10-26T04:03:15.107948Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def stopword_removal(text):\n    stops = set(stopwords.words(\"english\"))\n    return [word for word in text if not word in stops]","metadata":{"execution":{"iopub.status.busy":"2022-10-26T04:03:15.110720Z","iopub.execute_input":"2022-10-26T04:03:15.111371Z","iopub.status.idle":"2022-10-26T04:03:15.118389Z","shell.execute_reply.started":"2022-10-26T04:03:15.111329Z","shell.execute_reply":"2022-10-26T04:03:15.117252Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data['question_text'] = train_data['question_text'].apply(lambda x: stopword_removal(x))\ntrain_data.head()","metadata":{"execution":{"iopub.status.busy":"2022-10-26T04:03:15.120018Z","iopub.execute_input":"2022-10-26T04:03:15.120570Z","iopub.status.idle":"2022-10-26T04:06:10.457256Z","shell.execute_reply.started":"2022-10-26T04:03:15.120521Z","shell.execute_reply":"2022-10-26T04:06:10.456130Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from nltk.stem import WordNetLemmatizer\nwordnet_lemmatizer = WordNetLemmatizer()\ndef lemmatizer(text):\n    lemm_text = [wordnet_lemmatizer.lemmatize(word) for word in text]\n    return lemm_text","metadata":{"execution":{"iopub.status.busy":"2022-10-26T04:06:10.458680Z","iopub.execute_input":"2022-10-26T04:06:10.459272Z","iopub.status.idle":"2022-10-26T04:06:10.465509Z","shell.execute_reply.started":"2022-10-26T04:06:10.459233Z","shell.execute_reply":"2022-10-26T04:06:10.464546Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"nltk.download('omw-1.4')","metadata":{"execution":{"iopub.status.busy":"2022-10-26T04:06:10.466994Z","iopub.execute_input":"2022-10-26T04:06:10.468023Z","iopub.status.idle":"2022-10-26T04:06:10.772815Z","shell.execute_reply.started":"2022-10-26T04:06:10.467982Z","shell.execute_reply":"2022-10-26T04:06:10.771735Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data['question_text'] = train_data['question_text'].apply(lambda x: lemmatizer(x))\ntrain_data.head()","metadata":{"execution":{"iopub.status.busy":"2022-10-26T04:06:10.774180Z","iopub.execute_input":"2022-10-26T04:06:10.775192Z","iopub.status.idle":"2022-10-26T04:06:50.979559Z","shell.execute_reply.started":"2022-10-26T04:06:10.775152Z","shell.execute_reply":"2022-10-26T04:06:50.978415Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data['question_text'] = train_data['question_text'].apply(lambda x: join_words(x))\ntrain_data.head()","metadata":{"execution":{"iopub.status.busy":"2022-10-26T04:06:50.981557Z","iopub.execute_input":"2022-10-26T04:06:50.981962Z","iopub.status.idle":"2022-10-26T04:06:52.080913Z","shell.execute_reply.started":"2022-10-26T04:06:50.981924Z","shell.execute_reply":"2022-10-26T04:06:52.079888Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"insincere_data = train_data[train_data['target'] == 1]\nsincere_data = train_data[train_data['target'] == 0]","metadata":{"execution":{"iopub.status.busy":"2022-10-26T04:06:52.082780Z","iopub.execute_input":"2022-10-26T04:06:52.083320Z","iopub.status.idle":"2022-10-26T04:06:52.315279Z","shell.execute_reply.started":"2022-10-26T04:06:52.083276Z","shell.execute_reply":"2022-10-26T04:06:52.314233Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_top_n_non_stopwords(insincere_data['question_text'])","metadata":{"execution":{"iopub.status.busy":"2022-10-26T04:06:52.327458Z","iopub.execute_input":"2022-10-26T04:06:52.327772Z","iopub.status.idle":"2022-10-26T04:06:52.865982Z","shell.execute_reply.started":"2022-10-26T04:06:52.327744Z","shell.execute_reply":"2022-10-26T04:06:52.865064Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_top_n_non_stopwords(sincere_data['question_text'])","metadata":{"execution":{"iopub.status.busy":"2022-10-26T04:06:52.867497Z","iopub.execute_input":"2022-10-26T04:06:52.867864Z","iopub.status.idle":"2022-10-26T04:06:57.575877Z","shell.execute_reply.started":"2022-10-26T04:06:52.867815Z","shell.execute_reply":"2022-10-26T04:06:57.574936Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from nltk.util import ngrams\nfrom sklearn.feature_extraction.text import CountVectorizer\ndef get_top_ngram(corpus, n=None):\n    vec = CountVectorizer(ngram_range=(n,n)).fit(corpus)\n    bag_of_words = vec.transform(corpus)\n    sum_words = bag_of_words.sum(axis=0)\n    words_freq = [(word, sum_words[0, idx]) for word, idx in vec.vocabulary_.items()]\n    words_freq =sorted(words_freq, key = lambda x: x[1], reverse=True)\n    return words_freq[:10]","metadata":{"execution":{"iopub.status.busy":"2022-10-26T04:06:57.577454Z","iopub.execute_input":"2022-10-26T04:06:57.577820Z","iopub.status.idle":"2022-10-26T04:06:57.585378Z","shell.execute_reply.started":"2022-10-26T04:06:57.577784Z","shell.execute_reply":"2022-10-26T04:06:57.584234Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"top_n_bigrams = get_top_ngram(insincere_data['question_text'],2)[:10]\nx,y = map(list,zip(*top_n_bigrams))\nsns.barplot(x=y,y=x)","metadata":{"execution":{"iopub.status.busy":"2022-10-26T04:06:57.587095Z","iopub.execute_input":"2022-10-26T04:06:57.587461Z","iopub.status.idle":"2022-10-26T04:07:01.970106Z","shell.execute_reply.started":"2022-10-26T04:06:57.587427Z","shell.execute_reply":"2022-10-26T04:07:01.969163Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"top_n_trigrams = get_top_ngram(insincere_data['question_text'],3)[:10]\nx,y = map(list,zip(*top_n_trigrams))\nsns.barplot(x=y,y=x)","metadata":{"execution":{"iopub.status.busy":"2022-10-26T04:07:01.971659Z","iopub.execute_input":"2022-10-26T04:07:01.972027Z","iopub.status.idle":"2022-10-26T04:07:06.988116Z","shell.execute_reply.started":"2022-10-26T04:07:01.971990Z","shell.execute_reply":"2022-10-26T04:07:06.987153Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"top_n_bigrams_sincere = get_top_ngram(sincere_data['question_text'],2)[:10]\nx,y = map(list,zip(*top_n_bigrams_sincere))\nsns.barplot(x=y,y=x)","metadata":{"execution":{"iopub.status.busy":"2022-10-26T04:07:06.989548Z","iopub.execute_input":"2022-10-26T04:07:06.990293Z","iopub.status.idle":"2022-10-26T04:07:54.342114Z","shell.execute_reply.started":"2022-10-26T04:07:06.990250Z","shell.execute_reply":"2022-10-26T04:07:54.341146Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"top_n_trigrams_sincere = get_top_ngram(sincere_data['question_text'],3)[:10]\nx,y = map(list,zip(*top_n_trigrams_sincere))\nsns.barplot(x=y,y=x)","metadata":{"execution":{"iopub.status.busy":"2022-10-26T04:07:54.343455Z","iopub.execute_input":"2022-10-26T04:07:54.343812Z","iopub.status.idle":"2022-10-26T04:08:49.296120Z","shell.execute_reply.started":"2022-10-26T04:07:54.343774Z","shell.execute_reply":"2022-10-26T04:08:49.295193Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from textblob import TextBlob\ndef polarity(text):\n    return TextBlob(text).sentiment.polarity\ntrain_data['polarity_score'] = train_data['question_text'].apply(lambda x: polarity(x))\ntrain_data['polarity_score'].hist()","metadata":{"execution":{"iopub.status.busy":"2022-10-26T04:08:49.297661Z","iopub.execute_input":"2022-10-26T04:08:49.298025Z","iopub.status.idle":"2022-10-26T04:12:02.654215Z","shell.execute_reply.started":"2022-10-26T04:08:49.297988Z","shell.execute_reply":"2022-10-26T04:12:02.653127Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import spacy\nfrom spacy import displacy\nfrom collections import Counter\nimport en_core_web_sm\nnlp = en_core_web_sm.load()","metadata":{"execution":{"iopub.status.busy":"2022-10-26T04:12:02.655789Z","iopub.execute_input":"2022-10-26T04:12:02.656153Z","iopub.status.idle":"2022-10-26T04:12:13.360938Z","shell.execute_reply.started":"2022-10-26T04:12:02.656116Z","shell.execute_reply":"2022-10-26T04:12:13.359848Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from nltk.tokenize import WordPunctTokenizer\nfrom collections import Counter\nfrom string import punctuation, ascii_lowercase\nimport regex as re\nfrom tqdm import tqdm\n# setup tokenizer\ntokenizer = WordPunctTokenizer()\nvocab=Counter()\norg=Counter()\ngpe = Counter()\nstops = set(stopwords.words(\"english\"))\nlabels=[]\ndef process_ner(list_sentences):\n    for text in tqdm(list_sentences):\n        \n        doc = nlp(text)\n        for x in doc.ents:\n            if(x.label_=='PERSON'):\n                vocab.update([x.text.lower()])\n            if(x.label_=='ORG'):\n                org.update([x.text.lower()])\n            if(x.label_=='GPE'):\n                gpe.update([x.text.lower()])","metadata":{"execution":{"iopub.status.busy":"2022-10-26T04:12:13.362507Z","iopub.execute_input":"2022-10-26T04:12:13.363271Z","iopub.status.idle":"2022-10-26T04:12:13.384642Z","shell.execute_reply.started":"2022-10-26T04:12:13.363226Z","shell.execute_reply":"2022-10-26T04:12:13.383505Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"process_ner(insincere_data['question_text'])","metadata":{"execution":{"iopub.status.busy":"2022-10-26T04:12:13.386362Z","iopub.execute_input":"2022-10-26T04:12:13.387030Z","iopub.status.idle":"2022-10-26T04:20:42.416882Z","shell.execute_reply.started":"2022-10-26T04:12:13.386994Z","shell.execute_reply":"2022-10-26T04:20:42.415777Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(10,6))\nperson_most_common=pd.DataFrame(vocab.most_common(50))\nperson_most_common.columns=['Name','count']\npersonplot=sns.barplot(y=\"count\",x=\"Name\",data=person_most_common)\nloc, labels = plt.xticks(rotation='vertical')","metadata":{"execution":{"iopub.status.busy":"2022-10-26T04:20:42.418715Z","iopub.execute_input":"2022-10-26T04:20:42.419145Z","iopub.status.idle":"2022-10-26T04:20:43.338248Z","shell.execute_reply.started":"2022-10-26T04:20:42.419105Z","shell.execute_reply":"2022-10-26T04:20:43.337165Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(10,6))\norg_most_common=pd.DataFrame(org.most_common(50))\norg_most_common.columns=['Name','count']\norgplot=sns.barplot(y=\"count\",x=\"Name\",data=org_most_common)\nloc, labels = plt.xticks(rotation='vertical')","metadata":{"execution":{"iopub.status.busy":"2022-10-26T04:20:43.339912Z","iopub.execute_input":"2022-10-26T04:20:43.340338Z","iopub.status.idle":"2022-10-26T04:20:44.232273Z","shell.execute_reply.started":"2022-10-26T04:20:43.340291Z","shell.execute_reply":"2022-10-26T04:20:44.231192Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(10,6))\ngpe_most_common=pd.DataFrame(gpe.most_common(50))\ngpe_most_common.columns=['Name','count']\ngpeplot=sns.barplot(y=\"count\",x=\"Name\",data=gpe_most_common)\nloc, labels = plt.xticks(rotation='vertical')","metadata":{"execution":{"iopub.status.busy":"2022-10-26T04:20:44.233935Z","iopub.execute_input":"2022-10-26T04:20:44.234400Z","iopub.status.idle":"2022-10-26T04:20:45.066233Z","shell.execute_reply.started":"2022-10-26T04:20:44.234357Z","shell.execute_reply":"2022-10-26T04:20:45.065134Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data['question_text'] = test_data['question_text'].apply(lambda x: expand_contractions(x))\ntest_data.head()","metadata":{"execution":{"iopub.status.busy":"2022-10-26T04:20:45.067625Z","iopub.execute_input":"2022-10-26T04:20:45.068537Z","iopub.status.idle":"2022-10-26T04:21:02.330728Z","shell.execute_reply.started":"2022-10-26T04:20:45.068495Z","shell.execute_reply":"2022-10-26T04:21:02.329516Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data['question_text'] = test_data['question_text'].apply(lambda x: join_words(x))\ntest_data.head()","metadata":{"execution":{"iopub.status.busy":"2022-10-26T04:21:02.332514Z","iopub.execute_input":"2022-10-26T04:21:02.332924Z","iopub.status.idle":"2022-10-26T04:21:02.703030Z","shell.execute_reply.started":"2022-10-26T04:21:02.332885Z","shell.execute_reply":"2022-10-26T04:21:02.702073Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data = df_parallelize_run(test_data, text_clean_wrapper)","metadata":{"execution":{"iopub.status.busy":"2022-10-26T04:21:02.704476Z","iopub.execute_input":"2022-10-26T04:21:02.705106Z","iopub.status.idle":"2022-10-26T04:23:58.383261Z","shell.execute_reply.started":"2022-10-26T04:21:02.705063Z","shell.execute_reply":"2022-10-26T04:23:58.381650Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cur_vocabulary_test = set()\nfor text in tqdm(test_data['question_text'].values.tolist()):\n    words = text.split(' ')\n    cur_vocabulary_test.update(set(words))\nbug_punc_spacing_words_mapping = {}\nfor vocab in cur_vocabulary_test:\n    if '-' in vocab:\n        # whether the glove or para contain this word\n        if (vocab in embed_glove or vocab.capitalize() in embed_glove or vocab.lower() in embed_glove):\n            bug_punc_spacing_words_mapping[f\" {' - '.join(vocab.split('-'))} \"] = f\" {vocab} \"\n    \n    if '.' in vocab:\n        if vocab.endswith('.'):\n            continue\n        \n        if (vocab in embed_glove or vocab.capitalize() in embed_glove or vocab.lower() in embed_glove):\n            bug_punc_spacing_words_mapping[f\" {' . '.join(vocab.split('.'))} \"] = f\" {vocab} \"\n                                    \n# del bug_punc_spacing_words_mapping['  -  ']\nprint(f'found {len(bug_punc_spacing_words_mapping)} bug words')","metadata":{"execution":{"iopub.status.busy":"2022-10-26T04:23:58.385692Z","iopub.execute_input":"2022-10-26T04:23:58.386123Z","iopub.status.idle":"2022-10-26T04:23:59.647790Z","shell.execute_reply.started":"2022-10-26T04:23:58.386081Z","shell.execute_reply":"2022-10-26T04:23:59.646618Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data = df_parallelize_run(test_data, fix_dash_point_spacing_bug_wrapper)","metadata":{"execution":{"iopub.status.busy":"2022-10-26T04:23:59.649791Z","iopub.execute_input":"2022-10-26T04:23:59.650225Z","iopub.status.idle":"2022-10-26T04:28:47.019277Z","shell.execute_reply.started":"2022-10-26T04:23:59.650168Z","shell.execute_reply":"2022-10-26T04:28:47.017795Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data['question_text'] = test_data['question_text'].apply(lambda x: tokenization(x))\ntest_data.head()","metadata":{"execution":{"iopub.status.busy":"2022-10-26T04:28:47.021809Z","iopub.execute_input":"2022-10-26T04:28:47.022588Z","iopub.status.idle":"2022-10-26T04:29:34.586103Z","shell.execute_reply.started":"2022-10-26T04:28:47.022542Z","shell.execute_reply":"2022-10-26T04:29:34.585145Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data['question_text'] = test_data['question_text'].apply(lambda x: stopword_removal(x))\ntest_data.head()","metadata":{"execution":{"iopub.status.busy":"2022-10-26T04:29:34.587680Z","iopub.execute_input":"2022-10-26T04:29:34.587982Z","iopub.status.idle":"2022-10-26T04:30:24.953317Z","shell.execute_reply.started":"2022-10-26T04:29:34.587955Z","shell.execute_reply":"2022-10-26T04:30:24.952278Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data['question_text'] = test_data['question_text'].apply(lambda x: lemmatizer(x))\ntest_data.head()","metadata":{"execution":{"iopub.status.busy":"2022-10-26T04:30:24.955012Z","iopub.execute_input":"2022-10-26T04:30:24.955414Z","iopub.status.idle":"2022-10-26T04:30:37.132229Z","shell.execute_reply.started":"2022-10-26T04:30:24.955376Z","shell.execute_reply":"2022-10-26T04:30:37.131155Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data['question_text'] = test_data['question_text'].apply(lambda x: join_words(x))\ntest_data.head()","metadata":{"execution":{"iopub.status.busy":"2022-10-26T04:30:37.133895Z","iopub.execute_input":"2022-10-26T04:30:37.134365Z","iopub.status.idle":"2022-10-26T04:30:37.475977Z","shell.execute_reply.started":"2022-10-26T04:30:37.134297Z","shell.execute_reply":"2022-10-26T04:30:37.474780Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nfrom sklearn import metrics\n\nfrom keras.preprocessing.text import Tokenizer\nfrom keras.preprocessing.sequence import pad_sequences\nfrom keras.layers import Dense, Input, CuDNNLSTM, Embedding, Dropout, Activation, CuDNNGRU, Conv1D\nfrom keras.layers import Bidirectional, GlobalMaxPool1D, GlobalMaxPooling1D, GlobalAveragePooling1D\nfrom keras.layers import Input, Embedding, Dense, Conv2D, MaxPool2D, concatenate, Lambda\nfrom keras.layers import Reshape, Flatten, Concatenate, Dropout, SpatialDropout1D, BatchNormalization\nfrom tensorflow.keras.optimizers import Adam, Nadam\nfrom keras.models import Model\nfrom keras import backend as K\nfrom keras.callbacks import Callback\n\nfrom keras import initializers, regularizers, constraints, optimizers, layers\n\nfrom tensorflow.keras.preprocessing.text import Tokenizer\nfrom tensorflow.keras.preprocessing.sequence import pad_sequences\nfrom tensorflow.keras.callbacks import EarlyStopping, ModelCheckpoint, ReduceLROnPlateau\n\n# from tensorflow.keras.engine.topology import Layer\nfrom tensorflow.keras.layers import Layer\n\nimport gc, re\nfrom sklearn import metrics\nfrom sklearn.model_selection import GridSearchCV, StratifiedKFold\nfrom sklearn.metrics import f1_score, roc_auc_score, confusion_matrix, auc, precision_recall_curve","metadata":{"execution":{"iopub.status.busy":"2022-10-26T04:30:37.477530Z","iopub.execute_input":"2022-10-26T04:30:37.478183Z","iopub.status.idle":"2022-10-26T04:30:37.843867Z","shell.execute_reply.started":"2022-10-26T04:30:37.478142Z","shell.execute_reply":"2022-10-26T04:30:37.842843Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def load_glove(word_index):\n    EMBEDDING_FILE = '../input/glove-embed/glove.840B.300d.txt'\n\n    emb_mean,emb_std = -0.005838499,0.48782197\n    embed_size = 300\n\n    # word_index = tokenizer.word_index\n    nb_words = min(max_features, len(word_index))\n    #nb_words = len(word_index)\n    embedding_matrix = np.random.normal(emb_mean, emb_std, (nb_words, embed_size))\n    with open(EMBEDDING_FILE, 'r', encoding=\"utf8\") as f:\n        for line in f:\n            word, vec = line.split(' ', 1)\n            if word not in word_index:\n                continue\n            i = word_index[word]\n            if i >= nb_words:\n                continue\n            embedding_vector = np.asarray(vec.split(' '), dtype='float32')[:300]\n            if len(embedding_vector) == 300:\n                embedding_matrix[i] = embedding_vector\n    return embedding_matrix","metadata":{"execution":{"iopub.status.busy":"2022-10-26T04:30:37.845545Z","iopub.execute_input":"2022-10-26T04:30:37.845950Z","iopub.status.idle":"2022-10-26T04:30:37.854054Z","shell.execute_reply.started":"2022-10-26T04:30:37.845912Z","shell.execute_reply":"2022-10-26T04:30:37.853082Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_X = train_data[\"question_text\"].fillna(\"_##_\").values\nsplits = list(StratifiedKFold(n_splits=10,random_state=2018, shuffle=True).split(train_X,train_data['target'].values))\ntest_X = test_data[\"question_text\"].fillna(\"_##_\").values","metadata":{"execution":{"iopub.status.busy":"2022-10-26T04:30:37.855956Z","iopub.execute_input":"2022-10-26T04:30:37.856996Z","iopub.status.idle":"2022-10-26T04:30:38.316251Z","shell.execute_reply.started":"2022-10-26T04:30:37.856959Z","shell.execute_reply":"2022-10-26T04:30:38.315187Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_X, val_X = train_X[splits[0][0]],train_X[splits[0][1]]\ntrain_y = train_data['target'].values[splits[0][0]]\nval_y = train_data['target'].values[splits[0][1]]","metadata":{"execution":{"iopub.status.busy":"2022-10-26T04:30:38.318011Z","iopub.execute_input":"2022-10-26T04:30:38.318424Z","iopub.status.idle":"2022-10-26T04:30:38.376186Z","shell.execute_reply.started":"2022-10-26T04:30:38.318386Z","shell.execute_reply":"2022-10-26T04:30:38.375124Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"SEED = 2018\nembed_size = 300\nmax_features = None\nmaxlen = 57","metadata":{"execution":{"iopub.status.busy":"2022-10-26T04:30:38.378005Z","iopub.execute_input":"2022-10-26T04:30:38.378666Z","iopub.status.idle":"2022-10-26T04:30:38.383928Z","shell.execute_reply.started":"2022-10-26T04:30:38.378627Z","shell.execute_reply":"2022-10-26T04:30:38.382834Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tokenizer = Tokenizer(num_words=max_features, filters='', lower=True)\ntokenizer.fit_on_texts(list(train_X)+list(test_X))\ntrain_X = tokenizer.texts_to_sequences(train_X)\nval_X = tokenizer.texts_to_sequences(val_X)\ntest_X = tokenizer.texts_to_sequences(test_X)","metadata":{"execution":{"iopub.status.busy":"2022-10-26T04:30:38.386348Z","iopub.execute_input":"2022-10-26T04:30:38.387480Z","iopub.status.idle":"2022-10-26T04:31:05.460684Z","shell.execute_reply.started":"2022-10-26T04:30:38.387439Z","shell.execute_reply":"2022-10-26T04:31:05.459617Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_X = pad_sequences(train_X, maxlen=maxlen)\nval_X = pad_sequences(val_X, maxlen=maxlen)\ntest_X = pad_sequences(test_X, maxlen=maxlen)","metadata":{"execution":{"iopub.status.busy":"2022-10-26T04:31:05.462213Z","iopub.execute_input":"2022-10-26T04:31:05.462844Z","iopub.status.idle":"2022-10-26T04:31:10.173969Z","shell.execute_reply.started":"2022-10-26T04:31:05.462792Z","shell.execute_reply":"2022-10-26T04:31:10.172920Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"np.random.seed(SEED)\ntrn_idx = np.random.permutation(len(train_X))\nval_idx = np.random.permutation(len(val_X))","metadata":{"execution":{"iopub.status.busy":"2022-10-26T04:31:10.175641Z","iopub.execute_input":"2022-10-26T04:31:10.176030Z","iopub.status.idle":"2022-10-26T04:31:10.215131Z","shell.execute_reply.started":"2022-10-26T04:31:10.175990Z","shell.execute_reply":"2022-10-26T04:31:10.214086Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_X = train_X[trn_idx]\nval_X = val_X[val_idx]\ntrain_y = train_y[trn_idx]\nval_y = val_y[val_idx] ","metadata":{"execution":{"iopub.status.busy":"2022-10-26T04:31:10.216983Z","iopub.execute_input":"2022-10-26T04:31:10.217434Z","iopub.status.idle":"2022-10-26T04:31:10.677459Z","shell.execute_reply.started":"2022-10-26T04:31:10.217391Z","shell.execute_reply":"2022-10-26T04:31:10.676388Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def threshold_search(y_true, y_proba):\n    best_threshold = 0\n    best_score = 0\n    for threshold in [i * 0.01 for i in range(100)]:\n        score = f1_score(y_true=y_true, y_pred=y_proba > threshold)\n        if score > best_score:\n            best_threshold = threshold\n            best_score = score\n    rocauc = roc_auc_score(y_true, y_proba)\n    p, r, _ = precision_recall_curve(y_true, y_proba)\n    prauc = auc(r, p)\n    search_result = {'threshold': best_threshold, 'f1': best_score, 'rocauc': rocauc, 'prauc': prauc}\n    return search_result","metadata":{"execution":{"iopub.status.busy":"2022-10-26T04:31:10.679011Z","iopub.execute_input":"2022-10-26T04:31:10.679648Z","iopub.status.idle":"2022-10-26T04:31:10.686803Z","shell.execute_reply.started":"2022-10-26T04:31:10.679608Z","shell.execute_reply":"2022-10-26T04:31:10.685801Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class CyclicLR(Callback):\n    \"\"\"This callback implements a cyclical learning rate policy (CLR).\n    The method cycles the learning rate between two boundaries with\n    some constant frequency, as detailed in this paper (https://arxiv.org/abs/1506.01186).\n    The amplitude of the cycle can be scaled on a per-iteration or \n    per-cycle basis.\n    This class has three built-in policies, as put forth in the paper.\n    \"triangular\":\n        A basic triangular cycle w/ no amplitude scaling.\n    \"triangular2\":\n        A basic triangular cycle that scales initial amplitude by half each cycle.\n    \"exp_range\":\n        A cycle that scales initial amplitude by gamma**(cycle iterations) at each \n        cycle iteration.\n    For more detail, please see paper.\n    \n    # Example\n        ```python\n            clr = CyclicLR(base_lr=0.001, max_lr=0.006,\n                                step_size=2000., mode='triangular')\n            model.fit(X_train, Y_train, callbacks=[clr])\n        ```\n    Class also supports custom scaling functions:\n        ```python\n            clr_fn = lambda x: 0.5*(1+np.sin(x*np.pi/2.))\n            clr = CyclicLR(base_lr=0.001, max_lr=0.006,\n                                step_size=2000., scale_fn=clr_fn,\n                                scale_mode='cycle')\n            model.fit(X_train, Y_train, callbacks=[clr])\n        ```    \n    # Arguments\n        base_lr: initial learning rate which is the\n            lower boundary in the cycle.\n        max_lr: upper boundary in the cycle. Functionally,\n            it defines the cycle amplitude (max_lr - base_lr).\n            The lr at any cycle is the sum of base_lr\n            and some scaling of the amplitude; therefore \n            max_lr may not actually be reached depending on\n            scaling function.\n        step_size: number of training iterations per\n            half cycle. Authors suggest setting step_size\n            2-8 x training iterations in epoch.\n        mode: one of {triangular, triangular2, exp_range}.\n            Default 'triangular'.\n            Values correspond to policies detailed above.\n            If scale_fn is not None, this argument is ignored.\n        gamma: constant in 'exp_range' scaling function:\n            gamma**(cycle iterations)\n            scale_fn: Custom scaling policy defined by a single\n            argument lambda function, where \n            0 <= scale_fn(x) <= 1 for all x >= 0.\n            mode paramater is ignored \n        scale_mode: {'cycle', 'iterations'}.\n            Defines whether scale_fn is evaluated on \n            cycle number or cycle iterations (training\n            iterations since start of cycle). Default is 'cycle'.\n    \"\"\"\n\n    def __init__(self, base_lr=0.001, max_lr=0.006, step_size=2000., mode='triangular',\n                 gamma=1., scale_fn=None, scale_mode='cycle'):\n        super(CyclicLR, self).__init__()\n\n        self.base_lr = base_lr\n        self.max_lr = max_lr\n        self.step_size = step_size\n        self.mode = mode\n        self.gamma = gamma\n        if scale_fn == None:\n            if self.mode == 'triangular':\n                self.scale_fn = lambda x: 1.\n                self.scale_mode = 'cycle'\n            elif self.mode == 'triangular2':\n                self.scale_fn = lambda x: 1/(2.**(x-1))\n                self.scale_mode = 'cycle'\n            elif self.mode == 'exp_range':\n                self.scale_fn = lambda x: gamma**(x)\n                self.scale_mode = 'iterations'\n        else:\n            self.scale_fn = scale_fn\n            self.scale_mode = scale_mode\n        self.clr_iterations = 0.\n        self.trn_iterations = 0.\n        self.history = {}\n\n        self._reset()\n\n    def _reset(self, new_base_lr=None, new_max_lr=None,\n               new_step_size=None):\n        \"\"\"Resets cycle iterations.\n        Optional boundary/step size adjustment.\n        \"\"\"\n        if new_base_lr != None:\n            self.base_lr = new_base_lr\n        if new_max_lr != None:\n            self.max_lr = new_max_lr\n        if new_step_size != None:\n            self.step_size = new_step_size\n        self.clr_iterations = 0.\n        \n    def clr(self):\n        cycle = np.floor(1+self.clr_iterations/(2*self.step_size))\n        x = np.abs(self.clr_iterations/self.step_size - 2*cycle + 1)\n        if self.scale_mode == 'cycle':\n            return self.base_lr + (self.max_lr-self.base_lr)*np.maximum(0, (1-x))*self.scale_fn(cycle)\n        else:\n            return self.base_lr + (self.max_lr-self.base_lr)*np.maximum(0, (1-x))*self.scale_fn(self.clr_iterations)\n        \n    def on_train_begin(self, logs={}):\n        logs = logs or {}\n\n        if self.clr_iterations == 0:\n            K.set_value(self.model.optimizer.lr, self.base_lr)\n        else:\n            K.set_value(self.model.optimizer.lr, self.clr())        \n            \n    def on_batch_end(self, epoch, logs=None):\n        \n        logs = logs or {}\n        self.trn_iterations += 1\n        self.clr_iterations += 1\n\n        self.history.setdefault('lr', []).append(K.get_value(self.model.optimizer.lr))\n        self.history.setdefault('iterations', []).append(self.trn_iterations)\n        \n        for k, v in logs.items():\n            self.history.setdefault(k, []).append(v)\n        \n        K.set_value(self.model.optimizer.lr, self.clr())","metadata":{"execution":{"iopub.status.busy":"2022-10-26T04:31:10.688509Z","iopub.execute_input":"2022-10-26T04:31:10.689189Z","iopub.status.idle":"2022-10-26T04:31:10.708325Z","shell.execute_reply.started":"2022-10-26T04:31:10.689151Z","shell.execute_reply":"2022-10-26T04:31:10.707236Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# https://www.kaggle.com/qqgeogor/keras-lstm-attention-glove840b-lb-0-043\nclass Attention(Layer):\n    def __init__(self, step_dim,\n                 W_regularizer=None, b_regularizer=None,\n                 W_constraint=None, b_constraint=None,\n                 bias=True, **kwargs):\n        self.supports_masking = True\n        self.init = initializers.get('glorot_uniform')\n\n        self.W_regularizer = regularizers.get(W_regularizer)\n        self.b_regularizer = regularizers.get(b_regularizer)\n\n        self.W_constraint = constraints.get(W_constraint)\n        self.b_constraint = constraints.get(b_constraint)\n\n        self.bias = bias\n        self.step_dim = step_dim\n        self.features_dim = 0\n        super(Attention, self).__init__(**kwargs)\n\n    def build(self, input_shape):\n        assert len(input_shape) == 3\n\n        self.W = self.add_weight((input_shape[-1],),\n                                 initializer=self.init,\n                                 name='{}_W'.format(self.name),\n                                 regularizer=self.W_regularizer,\n                                 constraint=self.W_constraint)\n        self.features_dim = input_shape[-1]\n\n        if self.bias:\n            self.b = self.add_weight((input_shape[1],),\n                                     initializer='zero',\n                                     name='{}_b'.format(self.name),\n                                     regularizer=self.b_regularizer,\n                                     constraint=self.b_constraint)\n        else:\n            self.b = None\n\n        self.built = True\n\n    def compute_mask(self, input, input_mask=None):\n        return None\n\n    def call(self, x, mask=None):\n        features_dim = self.features_dim\n        step_dim = self.step_dim\n\n        eij = K.reshape(K.dot(K.reshape(x, (-1, features_dim)),\n                        K.reshape(self.W, (features_dim, 1))), (-1, step_dim))\n\n        if self.bias:\n            eij += self.b\n\n        eij = K.tanh(eij)\n        a = K.exp(eij)\n\n        if mask is not None:\n            a *= K.cast(mask, K.floatx())\n\n        a /= K.cast(K.sum(a, axis=1, keepdims=True) + K.epsilon(), K.floatx())\n\n        a = K.expand_dims(a)\n        weighted_input = x * a\n        return K.sum(weighted_input, axis=1)\n\n    def compute_output_shape(self, input_shape):\n        return input_shape[0],  self.features_dim","metadata":{"execution":{"iopub.status.busy":"2022-10-26T04:31:10.710091Z","iopub.execute_input":"2022-10-26T04:31:10.710545Z","iopub.status.idle":"2022-10-26T04:31:10.729329Z","shell.execute_reply.started":"2022-10-26T04:31:10.710503Z","shell.execute_reply":"2022-10-26T04:31:10.728206Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def model_lstm_atten(embedding_matrix):\n    inp = Input(shape=(maxlen,))\n    x = Embedding(max_features, embed_size, weights=[embedding_matrix], trainable=False)(inp)\n    x = SpatialDropout1D(0.2)(x)\n    x0 = Bidirectional(CuDNNLSTM(128, return_sequences=True))(x)\n#     x1 = Bidirectional(CuDNNGRU(64, return_sequences=True))(x0)\n    x2 = Bidirectional(CuDNNGRU(96, return_sequences=True))(x0)\n#     x3 = Attention(maxlen)(x2)\n#     x2 = CuDNNGRU(64, return_sequences=True)(x1)\n    y2 = GlobalMaxPooling1D()(x2)\n#     x = Concatenate()([y2, y1])\n#     y2 = BatchNormalization()(y2)\n    y2 = Dropout(0.1)(y2)\n    x = Dense(1, activation=\"sigmoid\")(y2)\n    model = Model(inputs=inp, outputs=x)\n    model.compile(loss='binary_crossentropy', optimizer=Adam(), metrics=['accuracy'])\n#     model.compile(loss='binary_crossentropy', optimizer=Nadam(), metrics=['accuracy'])\n    print(model.summary())\n    \n    return model","metadata":{"execution":{"iopub.status.busy":"2022-10-26T04:31:10.731061Z","iopub.execute_input":"2022-10-26T04:31:10.732584Z","iopub.status.idle":"2022-10-26T04:31:10.742561Z","shell.execute_reply.started":"2022-10-26T04:31:10.732538Z","shell.execute_reply":"2022-10-26T04:31:10.741421Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_train_list(train_X):\n    return [np.concatenate((np.ones((np.shape(train_X)[0],1))*max_features+1,train_X[:,1:]),1),train_X,np.concatenate((np.ones((np.shape(train_X)[0],1))*max_features+1,train_X[:,::-1][:,1:]),1)]","metadata":{"execution":{"iopub.status.busy":"2022-10-26T04:31:10.746277Z","iopub.execute_input":"2022-10-26T04:31:10.746685Z","iopub.status.idle":"2022-10-26T04:31:10.754739Z","shell.execute_reply.started":"2022-10-26T04:31:10.746627Z","shell.execute_reply":"2022-10-26T04:31:10.753764Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def train_pred(model, epochs=2):\n    for e in range(epochs):\n        model.fit(train_X, train_y, batch_size=512, epochs=1, validation_data=(val_X, val_y),verbose=1,callbacks=[clr])\n\n    pred_val_y = model.predict([val_X], batch_size=1024, verbose=0)\n    pred_test_y = model.predict([test_X], batch_size=1024, verbose=0)\n    return pred_val_y, pred_test_y","metadata":{"execution":{"iopub.status.busy":"2022-10-26T04:31:10.756152Z","iopub.execute_input":"2022-10-26T04:31:10.756799Z","iopub.status.idle":"2022-10-26T04:31:10.765777Z","shell.execute_reply.started":"2022-10-26T04:31:10.756722Z","shell.execute_reply":"2022-10-26T04:31:10.764832Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"word_index = tokenizer.word_index","metadata":{"execution":{"iopub.status.busy":"2022-10-26T04:31:10.769032Z","iopub.execute_input":"2022-10-26T04:31:10.769351Z","iopub.status.idle":"2022-10-26T04:31:10.776173Z","shell.execute_reply.started":"2022-10-26T04:31:10.769321Z","shell.execute_reply":"2022-10-26T04:31:10.775253Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"max_features = len(word_index)","metadata":{"execution":{"iopub.status.busy":"2022-10-26T04:31:10.777706Z","iopub.execute_input":"2022-10-26T04:31:10.778105Z","iopub.status.idle":"2022-10-26T04:31:10.785139Z","shell.execute_reply.started":"2022-10-26T04:31:10.778064Z","shell.execute_reply":"2022-10-26T04:31:10.784222Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"embedding_matrix = load_glove(word_index)\nnp.shape(embedding_matrix)","metadata":{"execution":{"iopub.status.busy":"2022-10-26T04:31:10.788418Z","iopub.execute_input":"2022-10-26T04:31:10.788754Z","iopub.status.idle":"2022-10-26T04:32:17.634295Z","shell.execute_reply.started":"2022-10-26T04:31:10.788728Z","shell.execute_reply":"2022-10-26T04:32:17.633170Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"outputs=[]","metadata":{"execution":{"iopub.status.busy":"2022-10-26T04:32:17.636675Z","iopub.execute_input":"2022-10-26T04:32:17.638108Z","iopub.status.idle":"2022-10-26T04:32:17.643750Z","shell.execute_reply.started":"2022-10-26T04:32:17.638035Z","shell.execute_reply":"2022-10-26T04:32:17.642671Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"clr = CyclicLR(base_lr=0.001, max_lr=0.003,step_size=300., mode='exp_range', gamma=0.99994)","metadata":{"execution":{"iopub.status.busy":"2022-10-26T04:32:17.645494Z","iopub.execute_input":"2022-10-26T04:32:17.646078Z","iopub.status.idle":"2022-10-26T04:32:17.654474Z","shell.execute_reply.started":"2022-10-26T04:32:17.646034Z","shell.execute_reply":"2022-10-26T04:32:17.653480Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred_val_y, pred_test_y = train_pred(model_lstm_atten(embedding_matrix), epochs = 4)\noutputs.append([pred_val_y, pred_test_y, 'LSTM w/ max'])\nresults = threshold_search(val_y, pred_val_y)\nprint(results)\nprint(confusion_matrix(val_y,pred_val_y>results['threshold']))","metadata":{"execution":{"iopub.status.busy":"2022-10-26T04:32:17.656415Z","iopub.execute_input":"2022-10-26T04:32:17.656909Z","iopub.status.idle":"2022-10-26T04:40:28.756715Z","shell.execute_reply.started":"2022-10-26T04:32:17.656869Z","shell.execute_reply":"2022-10-26T04:40:28.755540Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred_test_y","metadata":{"execution":{"iopub.status.busy":"2022-10-26T04:40:28.760187Z","iopub.execute_input":"2022-10-26T04:40:28.761500Z","iopub.status.idle":"2022-10-26T04:40:28.770304Z","shell.execute_reply.started":"2022-10-26T04:40:28.761456Z","shell.execute_reply":"2022-10-26T04:40:28.769189Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}