{"cells":[{"metadata":{"_uuid":"e94827fb932bf7407affe2d32b42b683a1c89ec3"},"cell_type":"markdown","source":"# Improve your Score with some Text Preprocessing\n\n## Updated version : \n > ###  https://www.kaggle.com/theoviel/improve-your-score-with-text-preprocessing-v2\n\n\n\n\nThis kernel is an improved version of @Dieter's work.\n> https://www.kaggle.com/christofhenkel/how-to-preprocessing-when-using-embeddings\n\n# Baically,the above preprocessing is now implemented and use for ZINDI platform (just like Kaggle) on a  competition named \"Basic Needs Basic Rights Kenya - Tech4MentalHealth\". \n\n**A big thanks to Dieter's work once again for this wonderful text preprocessing techniques!!!**\n\n**Aim and Objectives**: Is to classify text from university students in Kenya towards a mental health chatbot.\n\n\nIt is the preprocessing I use for my current LB score, and it has helped improving it by a bit. Feel free to use it as well, but please upvote if you do. \n\nThis is also how I caught a glimpse of spelling mistakes in the database.\n\n#### Any feedback is appreciated ! ","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 5GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b378958a9606ac48fe0dc54e24bed4cd503e0ac7"},"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport operator \nimport re","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"6e432bb170a329c79f58f78527bbbe4e857b2c41"},"cell_type":"markdown","source":"## Loading data","execution_count":null},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"train = pd.read_csv(\"/kaggle/input/train-data/Train_BNBR.csv\") #.drop('label', axis=1)\ntest = pd.read_csv(\"/kaggle/input/health-data/Test_health.csv\")\nlength = len(train)\ndf = pd.concat([train ,test])\n\nprint(\"Number of texts: \", df.shape[0])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_data = df[:length]\ntest_data = df[length:]","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"a453d10106aea7f6bf3f9bd952f2e02dd7a7472c"},"cell_type":"markdown","source":"## Loading embeddings","execution_count":null},{"metadata":{"trusted":true,"_uuid":"807e734e0ce617480c824f8bf26f0672f38397d5"},"cell_type":"code","source":"def load_embed(file):\n    def get_coefs(word,*arr): \n        return word, np.asarray(arr, dtype='float32')\n    \n    if file == '/kaggle/input/glove6b/glove.6B.300d.txt':\n        embeddings_index = dict(get_coefs(*o.split(\" \")) for o in open(file) if len(o)>100)\n    else:\n        embeddings_index = dict(get_coefs(*o.split(\" \")) for o in open(file, encoding='latin'))\n        \n    return embeddings_index","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"44662473bad36713bbe610e84039145c81e941b2"},"cell_type":"code","source":"glove = '/kaggle/input/glove6b/glove.6B.300d.txt'\nparagram =  '/kaggle/input/paragram-emb/paragram_300_sl999/paragram_300_sl999.txt'\nwiki_news = '/kaggle/input/fastext-emb/crawl-300d-2M.vec'\n#'/kaggle/input/wikinews/wiki-news-300d-1M1.vec'","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"7a0a98920daa420bf37a910739cda6bc716ba872"},"cell_type":"code","source":"print(\"Extracting GloVe embedding\")\nembed_glove = load_embed(glove)\nprint(\"Extracting Paragram embedding\")\nembed_paragram = load_embed(paragram)\nprint(\"Extracting FastText embedding\")\nembed_fasttext = load_embed(wiki_news)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"b5f9048e53a5cb91bdaa518d33fda756f1038562"},"cell_type":"markdown","source":"## Vocabulary and Coverage functions\n> Again, check Dieter's work if you haven't, those are his.","execution_count":null},{"metadata":{"trusted":true,"_uuid":"c362fc1868aaa53dfc0bd91d453419815eba7c47"},"cell_type":"code","source":"def build_vocab(texts):\n    sentences = texts.apply(lambda x: x.split()).values\n    vocab = {}\n    for sentence in sentences:\n        for word in sentence:\n            try:\n                vocab[word] += 1\n            except KeyError:\n                vocab[word] = 1\n    return vocab","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f35a7213fc9a7e80a7c210d11b3a8094d3a8e07e"},"cell_type":"code","source":"def check_coverage(vocab, embeddings_index):\n    known_words = {}\n    unknown_words = {}\n    nb_known_words = 0\n    nb_unknown_words = 0\n    for word in vocab.keys():\n        try:\n            known_words[word] = embeddings_index[word]\n            nb_known_words += vocab[word]\n        except:\n            unknown_words[word] = vocab[word]\n            nb_unknown_words += vocab[word]\n            pass\n\n    print('Found embeddings for {:.2%} of vocab'.format(len(known_words) / len(vocab)))\n    print('Found embeddings for  {:.2%} of all text'.format(nb_known_words / (nb_known_words + nb_unknown_words)))\n    unknown_words = sorted(unknown_words.items(), key=operator.itemgetter(1))[::-1]\n\n    return unknown_words","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"31a3aff72321746bdf6731236d183ee28931793b"},"cell_type":"markdown","source":"## Starting point","execution_count":null},{"metadata":{"trusted":true,"_uuid":"cbf1970d63ea82cc32817b6d2e5f2b976276a08f"},"cell_type":"code","source":"vocab = build_vocab(df['text'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"42c2b82740ac82f5678a46b7c82b5525616c1304","scrolled":true},"cell_type":"code","source":"print(\"Glove : \")\noov_glove = check_coverage(vocab, embed_glove)\nprint(\"Paragram : \")\noov_paragram = check_coverage(vocab, embed_paragram)\nprint(\"FastText : \")\noov_fasttext = check_coverage(vocab, embed_fasttext)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"7b32cdaef54a0e7b428d1b82ee53afec9f5dd49b"},"cell_type":"markdown","source":" #### Paragram seems to have a significantly lower coverage. \n>That's because it does not understand upper letters, let us lower our texts :","execution_count":null},{"metadata":{"trusted":true,"_uuid":"c657fb61a5942ee9807258364be3a2d1058471e6"},"cell_type":"code","source":"df['lowered_text'] = df['text'].apply(lambda x: x.lower())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a3c70ee2f7670eb6278a855130e5d2426238f590"},"cell_type":"code","source":"vocab_low = build_vocab(df['lowered_text'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"6e4f5244153ac68b38213a36c3434b3c3a2a2441"},"cell_type":"code","source":"print(\"Glove : \")\noov_glove = check_coverage(vocab_low, embed_glove)\nprint(\"Paragram : \")\noov_paragram = check_coverage(vocab_low, embed_paragram)\nprint(\"FastText : \")\noov_fasttext = check_coverage(vocab_low, embed_fasttext)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"5d0ac02d8aec1460f3bdc19f6db815fcb84443a8"},"cell_type":"markdown","source":"#### Better, but we lost a bit of information on the other embeddings.\n> Therer are words known that are known with upper letters and unknown without. Let us fix that :\n- word.lower() takes the embedding of word if word.lower() doesn't have an embedding","execution_count":null},{"metadata":{"trusted":true,"_uuid":"7703322c717121d6f8ee06ab7316232c75448ebe"},"cell_type":"code","source":"def add_lower(embedding, vocab):\n    count = 0\n    for word in vocab:\n        if word in embedding and word.lower() not in embedding:  \n            embedding[word.lower()] = embedding[word]\n            count += 1\n    print(f\"Added {count} words to embedding\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"77757cf80eeda36376dca64737c61d3100c1acf4"},"cell_type":"code","source":"print(\"Glove : \")\nadd_lower(embed_glove, vocab)\nprint(\"Paragram : \")\nadd_lower(embed_paragram, vocab)\nprint(\"FastText : \")\nadd_lower(embed_fasttext, vocab)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"1dec1b447767d99aa4814378b85365202e6bc230"},"cell_type":"code","source":"print(\"Glove : \")\noov_glove = check_coverage(vocab_low, embed_glove)\nprint(\"Paragram : \")\noov_paragram = check_coverage(vocab_low, embed_paragram)\nprint(\"FastText : \")\noov_fasttext = check_coverage(vocab_low, embed_fasttext)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"0cd0a201ddd3558ece968bbede941e7f7aee386d"},"cell_type":"markdown","source":"### What's wrong ?","execution_count":null},{"metadata":{"trusted":true,"_uuid":"68ee5f7fdf4def1aa67f76144ed4df9380087619"},"cell_type":"code","source":"oov_glove[:100]","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"768c3eca0b5342473dd9c9e208eb90e82eb13328"},"cell_type":"markdown","source":"#### First faults appearing are : \n- Contractions \n- Words with punctuation in them\n\n> Let us correct that.","execution_count":null},{"metadata":{"_uuid":"3849b08b9edb1aea040dd61c9fa3f2bb1a58482a"},"cell_type":"markdown","source":"## Contractions","execution_count":null},{"metadata":{"trusted":true,"_uuid":"fb3eb7d653a20b4d8157b08de690b49dfb0148eb"},"cell_type":"code","source":"contraction_mapping = {\"don't\":\"do not\",\"ain't\": \"is not\", \"aren't\": \"are not\",\"can't\": \"cannot\", \"cause\": \"because\", \"could've\": \"could have\", \"couldn't\": \"could not\", \"didn't\": \"did not\",  \"doesn't\": \"does not\", \"don't\": \"do not\", \"hadn't\": \"had not\", \"hasn't\": \"has not\", \"haven't\": \"have not\", \"he'd\": \"he would\",\"he'll\": \"he will\", \"he's\": \"he is\", \"how'd\": \"how did\", \"how'd'y\": \"how do you\", \"how'll\": \"how will\", \"how's\": \"how is\",  \"I'd\": \"I would\", \"I'd've\": \"I would have\", \"I'll\": \"I will\", \"I'll've\": \"I will have\",\"I'm\": \"I am\", \"I've\": \"I have\", \"i'd\": \"i would\", \"i'd've\": \"i would have\", \"i'll\": \"i will\",  \"i'll've\": \"i will have\",\"i'm\": \"i am\", \"i've\": \"i have\", \"isn't\": \"is not\", \"it'd\": \"it would\", \"it'd've\": \"it would have\", \"it'll\": \"it will\", \"it'll've\": \"it will have\",\"it's\": \"it is\", \"let's\": \"let us\", \"ma'am\": \"madam\", \"mayn't\": \"may not\", \"might've\": \"might have\",\"mightn't\": \"might not\",\"mightn't've\": \"might not have\", \"must've\": \"must have\", \"mustn't\": \"must not\", \"mustn't've\": \"must not have\", \"needn't\": \"need not\", \"needn't've\": \"need not have\",\"o'clock\": \"of the clock\", \"oughtn't\": \"ought not\", \"oughtn't've\": \"ought not have\", \"shan't\": \"shall not\", \"sha'n't\": \"shall not\", \"shan't've\": \"shall not have\", \"she'd\": \"she would\", \"she'd've\": \"she would have\", \"she'll\": \"she will\", \"she'll've\": \"she will have\", \"she's\": \"she is\", \"should've\": \"should have\", \"shouldn't\": \"should not\", \"shouldn't've\": \"should not have\", \"so've\": \"so have\",\"so's\": \"so as\", \"this's\": \"this is\",\"that'd\": \"that would\", \"that'd've\": \"that would have\", \"that's\": \"that is\", \"there'd\": \"there would\", \"there'd've\": \"there would have\", \"there's\": \"there is\", \"here's\": \"here is\",\"they'd\": \"they would\", \"they'd've\": \"they would have\", \"they'll\": \"they will\", \"they'll've\": \"they will have\", \"they're\": \"they are\", \"they've\": \"they have\", \"to've\": \"to have\", \"wasn't\": \"was not\", \"we'd\": \"we would\", \"we'd've\": \"we would have\", \"we'll\": \"we will\", \"we'll've\": \"we will have\", \"we're\": \"we are\", \"we've\": \"we have\", \"weren't\": \"were not\", \"what'll\": \"what will\", \"what'll've\": \"what will have\", \"what're\": \"what are\",  \"what's\": \"what is\", \"what've\": \"what have\", \"when's\": \"when is\", \"when've\": \"when have\", \"where'd\": \"where did\", \"where's\": \"where is\", \"where've\": \"where have\", \"who'll\": \"who will\", \"who'll've\": \"who will have\", \"who's\": \"who is\", \"who've\": \"who have\", \"why's\": \"why is\", \"why've\": \"why have\", \"will've\": \"will have\", \"won't\": \"will not\", \"won't've\": \"will not have\", \"would've\": \"would have\", \"wouldn't\": \"would not\", \"wouldn't've\": \"would not have\", \"y'all\": \"you all\", \"y'all'd\": \"you all would\",\"y'all'd've\": \"you all would have\",\"y'all're\": \"you all are\",\"y'all've\": \"you all have\",\"you'd\": \"you would\", \"you'd've\": \"you would have\", \"you'll\": \"you will\", \"you'll've\": \"you will have\", \"you're\": \"you are\", \"you've\": \"you have\" }","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"348a75ce1e902dd34eff964b6a511310504fcfa6"},"cell_type":"code","source":"def known_contractions(embed):\n    known = []\n    for contract in contraction_mapping:\n        if contract in embed:\n            known.append(contract)\n    return known","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e9e43ac139947e32fd3a9f357ebbe1110a56d875"},"cell_type":"code","source":"print(\"- Known Contractions -\")\nprint(\"   Glove :\")\nprint(known_contractions(embed_glove))\nprint(\"   Paragram :\")\nprint(known_contractions(embed_paragram))\nprint(\"   FastText :\")\nprint(known_contractions(embed_fasttext))","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"a082b84860096732f927b97c421bb0a4c6579ca2"},"cell_type":"markdown","source":"#### FastText does not understand contractions\n> We use the map to replace them","execution_count":null},{"metadata":{"trusted":true,"_uuid":"d7d6f63b2da86b33d2650114630c57d47f3cc567"},"cell_type":"code","source":"def clean_contractions(text, mapping):\n    specials = [\"’\", \"‘\", \"´\", \"`\"]\n    for s in specials:\n        text = text.replace(s, \"'\")\n    text = ' '.join([mapping[t] if t in mapping else t for t in text.split(\" \")])\n    return text","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"eeebeb5d860a0d5c5d8c9cdfc30ed652cea17634"},"cell_type":"code","source":"df['treated_text'] = df['lowered_text'].apply(lambda x: clean_contractions(x, contraction_mapping))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"490030c049c9c5d32e892da02816de053f8a252e"},"cell_type":"code","source":"vocab = build_vocab(df['treated_text'])\nprint(\"Glove : \")\noov_glove = check_coverage(vocab, embed_glove)\nprint(\"Paragram : \")\noov_paragram = check_coverage(vocab, embed_paragram)\nprint(\"FastText : \")\noov_fasttext = check_coverage(vocab, embed_fasttext)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"dba213d3a9fefce81a3ab41e2ba95846f87da997"},"cell_type":"markdown","source":"## Now, let us deal with special characters","execution_count":null},{"metadata":{"trusted":true,"_uuid":"70f766c527f6b7d87a0213306d20db49a5cc014f"},"cell_type":"code","source":"punct = \"/-'?!.,#$%\\'()*+-/:;<=>@[\\\\]^_`{|}~\" + '\"\"“”’' + '∞θ÷α•à−β∅³π‘₹´°£€\\×™√²—–&â¦™'    ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c95828228f2b28c55a02890c5a882037cad18262"},"cell_type":"code","source":"def unknown_punct(embed, punct):\n    unknown = ''\n    for p in punct:\n        if p not in embed:\n            unknown += p\n            unknown += ' '\n    return unknown","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"69129cad1f08fda5912051d273db012f9adb081a"},"cell_type":"code","source":"print(\"Glove :\")\nprint(unknown_punct(embed_glove, punct))\nprint(\"Paragram :\")\nprint(unknown_punct(embed_paragram, punct))\nprint(\"FastText :\")\nprint(unknown_punct(embed_fasttext, punct))","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"b0b4dc88770ed22b650dcd590a343f8169dbd9d2"},"cell_type":"markdown","source":"#### FastText seems to have a better knowledge of special characters \n> We use a map to replace unknown characters with known ones.\n\n> We make sure there are spaces between words and punctuation\n","execution_count":null},{"metadata":{"trusted":true,"_uuid":"2ebe824ff42bc0f93aca2884b430cc96c12857b8"},"cell_type":"code","source":"punct_mapping = {\"‘\": \"'\", \"₹\": \"e\", \"´\": \"'\", \"°\": \"\", \"€\": \"e\", \"™\": \"tm\", \"√\": \" sqrt \", \"×\": \"x\", \"²\": \"2\", \"—\": \"-\", \"–\": \"-\", \"’\": \"'\", \"_\": \"-\", \"`\": \"'\", '“': '\"', '”': '\"', '“': '\"', \"£\": \"e\", '∞': 'infinity', 'θ': 'theta', '÷': '/', 'α': 'alpha', '•': '.', 'à': 'a', '−': '-', 'β': 'beta', '∅': '', '³': '3', 'π': 'pi'}","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a3825a0af80d6443e6e2665edadf2c394d8c92ff"},"cell_type":"code","source":"def clean_special_chars(text, punct, mapping):\n    for p in mapping:\n        text = text.replace(p, mapping[p])\n    \n    for p in punct:\n        text = text.replace(p, f' {p} ')\n    \n    specials = {'\\u200b': ' ', '…': ' ... ', '\\ufeff': '', 'करना': '', 'है': ''}  # Other special characters that I have to deal with in last\n    for s in specials:\n        text = text.replace(s, specials[s])\n    \n    return text","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b47c945e6f02c2a9f72aedd72da4681695cb20dd"},"cell_type":"code","source":"df['treated_text'] = df['treated_text'].apply(lambda x: clean_special_chars(x, punct, punct_mapping))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"1d89484c4cd04be56e6d81ea6b2b9ded41371ef8"},"cell_type":"code","source":"vocab = build_vocab(df['treated_text'])\nprint(\"Glove : \")\noov_glove = check_coverage(vocab, embed_glove)\nprint(\"Paragram : \")\noov_paragram = check_coverage(vocab, embed_paragram)\nprint(\"FastText : \")\noov_fasttext = check_coverage(vocab, embed_fasttext)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"4d206d53f696e7d9ee2f7463c8514c2f3bf6ce15"},"cell_type":"code","source":"oov_fasttext[:900]","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"63a73741168e19b59ff8efaf167e2fe113725d7e"},"cell_type":"markdown","source":"### What's still missing ? \n- Unknown words\n- Acronyms\n- Spelling mistakes","execution_count":null},{"metadata":{"_uuid":"6be1bab080b82dfbc131fb77c0beecb5767263d3"},"cell_type":"markdown","source":"## We can correct manually most frequent mispells","execution_count":null},{"metadata":{"_uuid":"2fd418d76ecf3f4e9264c684e728bf26aafc4237"},"cell_type":"markdown","source":"#### For example, here are some mistakes and their frequency\n- lowi : 2 times\n- havei : 2 times\n- psycologically : 1 time and so on like that.","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"mispell_dict = {'dieam': 'die',\n                'stresseed':'stressed',\n                'lowi': 'low i',\n                'havei':'have i',\n                'worthlesness':'worthlessness',\n                'nightsi':'nights i',\n                'downrecovering':'down recovering',\n                'deferrin':'deferring',\n                'helplessi':'helpless i',\n                'foind':'found',\n                'lowt':'low',\n                'helplessstill':'helpless still',\n                'birthnow':'birth now',\n                'patners':'partners',\n                'motivationsuicidal':'motivation suicidal',\n                'includingmy':'including my',\n                'hopelessnesss':'hopelessness',\n                'desserted':'deserted',\n                'dillusioned':'disllusioned',\n                'issolated':'isolated',\n                'undestands':'understands',\n                'oldself':'old self',\n                'ponographic':'pornographic',\n                'unworthyness':'unworthiness',\n                'selfworth':'self worth',\n                'worldm':'world',\n                'incidencesof':'incidences of',\n                'weathernow': 'weather now',\n                'benefitto':'benefit to',\n                'frustratedi':'frustrated i',\n                'worhtless':'worthless',\n                'doto':'do to',\n                'messnow':'mess now',\n                'negativecurrently':'negative currently',\n                'schoolfee':'school fee',\n                'existd':'existed',\n                'lonelycurrently':'lonely currently',\n                'diserted':'deserted',\n                'schoolfees':'school fees',\n                'drinnking':'drinking',\n                'alccohol':'alcohol',\n                'addidcted':'addicted',\n                'depressioni':'depression i',\n                'childlish':'childish',\n                'lonelyi':'lonely i',\n                'stresssed':'stressed',\n                'deteroriating':'deteriorating',\n                'liferight':'life right',\n                'thingof':'thing of',\n                'whren':'when',\n                'everythingi':'everything i',\n                'insomia':'insomnia',\n                'dizzines':'dizziness',\n                'hopelessfor':'hopeless for',\n                'confusednow':'confused now',\n                'isolatednow':'isolated now',\n                'lonelynow':'lonely now',\n                'negletion':'neglection',\n                'sucidal':'suicidal',\n                'psycologically':'psychologically',\n                'avoiod':'avoid',\n                'cornerfeeling':'corner feeling',\n                'hatrednow':'hatred now',\n                'mediataton':'medication',\n                'frorge':'forget',\n                'drugsnow':'drugs now'}","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"2bdc15dd0f50de811510fd870fcaae9c137c42c3"},"cell_type":"code","source":"def correct_spelling(x, dic):\n    for word in dic.keys():\n        x = x.replace(word, dic[word])\n    return x","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c887a0a6f7498724a2c377e9e29fac683ea3a59b"},"cell_type":"code","source":"df['treated_text'] = df['treated_text'].apply(lambda x: correct_spelling(x, mispell_dict))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"63aba43671245f1ebc87540a8bd49a174791c378"},"cell_type":"code","source":"vocab = build_vocab(df['treated_text'])\nprint(\"Glove : \")\noov_glove = check_coverage(vocab, embed_glove)\nprint(\"Paragram : \")\noov_paragram = check_coverage(vocab, embed_paragram)\nprint(\"FastText : \")\noov_fasttext = check_coverage(vocab, embed_fasttext)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"a7bd1d4908fdcb485434c056a72bdd04555bf91d"},"cell_type":"markdown","source":"# Wow....All embeddings now near 100% accuracy\n\n# Nice one !","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"df.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_data = df[:length]\ntest_data = df[length:]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_data.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test_data.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_data = train_data.drop(['text', 'lowered_text'], axis =1)\ntrain_data.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test_data = test_data.drop(['label','text','lowered_text'], axis =1)\ntest_data.head()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Let's now save the Preprocessed datasets","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"train_data.to_csv(\"cleaned(925)_train.csv\", index = False)\ntest_data.to_csv(\"cleaned_(925)test.csv\", index =False)","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}