{"cells":[{"metadata":{"trusted":true,"_uuid":"b378958a9606ac48fe0dc54e24bed4cd503e0ac7"},"cell_type":"code","source":"import pandas as pd\nfrom tqdm import tqdm\ntqdm.pandas()\nimport sys\nimport regex as re\nfrom spacy.lang.en import English\n\nimport os\nimport gc\nimport re\n\nfrom tensorflow.keras.preprocessing.text import Tokenizer\nfrom tensorflow.keras.preprocessing.sequence import pad_sequences","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"6a3ca3fdd0c15565ccb55809038524220042fccf"},"cell_type":"code","source":"SEED = 1029","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"3e050f2fa9668c765466d766cfd79720cf6cc819"},"cell_type":"code","source":"def build_vocab(sentences, verbose =  True):\n    \"\"\"\n    :param sentences: list of list of words\n    :return: dictionary of words and their count\n    \"\"\"\n    vocab = {}\n    for sentence in tqdm(sentences, disable = (not verbose)):\n        for word in sentence:\n            try:\n                vocab[word] += 1\n            except KeyError:\n                vocab[word] = 1\n    return vocab","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a9a4ff66a05f94002cbe6ab6ac2e33fca15bc2ee"},"cell_type":"code","source":"puncts = [',', '.', '\"', ':', ')', '(', '-', '!', '?', '|', ';', \"'\", '$', '&', '/', '[', ']', '>', '%', '=', '#', '*', '+', '\\\\','•',  '~', '@', '£', \n '·', '_', '{', '}', '©', '^', '®', '`',  '<', '→', '°', '€', '™', '›',  '♥', '←', '×', '§', '″', '′', 'Â', '█', '½', 'à', '…', \n '“', '★', '”', '–', '●', 'â', '►', '−', '¢', '²', '¬', '░', '¶', '↑', '±', '¿', '▾', '═', '¦', '║', '―', '¥', '▓', '—', '‹', '─', \n '▒', '：', '¼', '⊕', '▼', '▪', '†', '■', '’', '▀', '¨', '▄', '♫', '☆', 'é', '¯', '♦', '¤', '▲', 'è', '¸', '¾', 'Ã', '⋅', '‘', '∞', \n '∙', '）', '↓', '、', '│', '（', '»', '，', '♪', '╩', '╚', '³', '・', '╦', '╣', '╔', '╗', '▬', '❤', 'ï', 'Ø', '¹', '≤', '‡', '√', ]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"68bb8dcb4e9be18be2ead10e56e5c6795d2e56a6"},"cell_type":"code","source":"puncts = [',', '.', '\"', ':', ')', '(', '-', '!', '?', '|', ';', \"'\", '$', '&', '/', '[', ']', '>', '%', '=', '#', '*', '+', '\\\\', '•',  '~', '@', '£', \n '·', '_', '{', '}', '©', '^', '®', '`',  '<', '→', '°', '€', '™', '›',  '♥', '←', '×', '§', '″', '′', 'Â', '█', '½', 'à', '…', \n '“', '★', '”', '–', '●', 'â', '►', '−', '¢', '²', '¬', '░', '¶', '↑', '±', '¿', '▾', '═', '¦', '║', '―', '¥', '▓', '—', '‹', '─', \n '▒', '：', '¼', '⊕', '▼', '▪', '†', '■', '’', '▀', '¨', '▄', '♫', '☆', 'é', '¯', '♦', '¤', '▲', 'è', '¸', '¾', 'Ã', '⋅', '‘', '∞', \n '∙', '）', '↓', '、', '│', '（', '»', '，', '♪', '╩', '╚', '³', '・', '╦', '╣', '╔', '╗', '▬', '❤', 'ï', 'Ø', '¹', '≤', '‡', '√', ]\n\ndef clean_text(x):\n    x = str(x)\n    for punct in puncts:\n        x = x.replace(punct, f' {punct} ')\n    return x\n\ndef clean_numbers(x):\n    x = re.sub('[0-9]{5,}', '#####', x)\n    x = re.sub('[0-9]{4}', '####', x)\n    x = re.sub('[0-9]{3}', '###', x)\n    x = re.sub('[0-9]{2}', '##', x)\n    return x\n\nmispell_dict = {\"aren't\" : \"are not\",\n\"can't\" : \"cannot\",\n\"couldn't\" : \"could not\",\n\"didn't\" : \"did not\",\n\"doesn't\" : \"does not\",\n\"don't\" : \"do not\",\n\"hadn't\" : \"had not\",\n\"hasn't\" : \"has not\",\n\"haven't\" : \"have not\",\n\"he'd\" : \"he would\",\n\"he'll\" : \"he will\",\n\"he's\" : \"he is\",\n\"i'd\" : \"I would\",\n\"i'd\" : \"I had\",\n\"i'll\" : \"I will\",\n\"i'm\" : \"I am\",\n\"isn't\" : \"is not\",\n\"it's\" : \"it is\",\n\"it'll\":\"it will\",\n\"i've\" : \"I have\",\n\"let's\" : \"let us\",\n\"mightn't\" : \"might not\",\n\"mustn't\" : \"must not\",\n\"shan't\" : \"shall not\",\n\"she'd\" : \"she would\",\n\"she'll\" : \"she will\",\n\"she's\" : \"she is\",\n\"shouldn't\" : \"should not\",\n\"that's\" : \"that is\",\n\"there's\" : \"there is\",\n\"they'd\" : \"they would\",\n\"they'll\" : \"they will\",\n\"they're\" : \"they are\",\n\"they've\" : \"they have\",\n\"we'd\" : \"we would\",\n\"we're\" : \"we are\",\n\"weren't\" : \"were not\",\n\"we've\" : \"we have\",\n\"what'll\" : \"what will\",\n\"what're\" : \"what are\",\n\"what's\" : \"what is\",\n\"what've\" : \"what have\",\n\"where's\" : \"where is\",\n\"who'd\" : \"who would\",\n\"who'll\" : \"who will\",\n\"who're\" : \"who are\",\n\"who's\" : \"who is\",\n\"who've\" : \"who have\",\n\"won't\" : \"will not\",\n\"wouldn't\" : \"would not\",\n\"you'd\" : \"you would\",\n\"you'll\" : \"you will\",\n\"you're\" : \"you are\",\n\"you've\" : \"you have\",\n\"'re\": \" are\",\n\"wasn't\": \"was not\",\n\"we'll\":\" will\",\n\"didn't\": \"did not\",\n\"tryin'\":\"trying\"}\n\ndef _get_mispell(mispell_dict):\n    mispell_re = re.compile('(%s)' % '|'.join(mispell_dict.keys()))\n    return mispell_dict, mispell_re\n\nmispellings, mispellings_re = _get_mispell(mispell_dict)\ndef replace_typical_misspell(text):\n    def replace(match):\n        return mispellings[match.group(0)]\n    return mispellings_re.sub(replace, text)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d471b161d66c85a990f9692ff67b44fed47d86d3"},"cell_type":"code","source":"def load_and_prec():\n    train_df = pd.read_csv(\"../input/train.csv\")[:1000]\n    test_df = pd.read_csv(\"../input/test.csv\")[:1000]\n    print(\"Train shape : \",train_df.shape)\n    print(\"Test shape : \",test_df.shape)\n    \n    # lower\n    train_df[\"question_text\"] = train_df[\"question_text\"].progress_apply(lambda x: x.lower())\n    test_df[\"question_text\"] = test_df[\"question_text\"].progress_apply(lambda x: x.lower())\n    \n    # Clean the text\n    train_df[\"question_text\"] = train_df[\"question_text\"].progress_apply(lambda x: clean_text(x))\n    test_df[\"question_text\"] = test_df[\"question_text\"].progress_apply(lambda x: clean_text(x))\n    \n    # Clean numbers\n    train_df[\"question_text\"] = train_df[\"question_text\"].progress_apply(lambda x: clean_numbers(x))\n    test_df[\"question_text\"] = test_df[\"question_text\"].progress_apply(lambda x: clean_numbers(x))\n    \n    # Clean speelings\n    train_df[\"question_text\"] = train_df[\"question_text\"].progress_apply(lambda x: replace_typical_misspell(x))\n    test_df[\"question_text\"] = test_df[\"question_text\"].progress_apply(lambda x: replace_typical_misspell(x))\n    \n    ## fill up the missing values\n    train_X = train_df[\"question_text\"].fillna(\"_##_\").values\n    test_X = test_df[\"question_text\"].fillna(\"_##_\").values\n\n    ## Tokenize the sentences\n    tokenizer = Tokenizer(num_words=max_features)\n    tokenizer.fit_on_texts(list(train_X))\n    train_X = tokenizer.texts_to_sequences(train_X)\n    test_X = tokenizer.texts_to_sequences(test_X)\n\n    ## Pad the sentences \n    train_X = pad_sequences(train_X, maxlen=maxlen)\n    test_X = pad_sequences(test_X, maxlen=maxlen)\n\n    ## Get the target values\n    train_y = train_df['target'].values\n    \n    #shuffling the data\n    np.random.seed(SEED)\n    trn_idx = np.random.permutation(len(train_X))\n\n    train_X = train_X[trn_idx]\n    train_y = train_y[trn_idx]\n    \n    return train_X, test_X, train_y, tokenizer.word_index","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true,"scrolled":true},"cell_type":"code","source":"train_sentences = train[\"question_text\"].progress_apply(lambda x: tokenize(x)).values\ntest_sentences = test[\"question_text\"].progress_apply(lambda x: tokenize(x)).values\n\ntrain_vocab = build_vocab(train_sentences)\ntest_vocab = build_vocab(test_sentences)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"2309b03c9f4458930a1f4eeee8c6fd132ea43876"},"cell_type":"code","source":"import torchtext","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"807e734e0ce617480c824f8bf26f0672f38397d5"},"cell_type":"code","source":"glove_vectors = torchtext.vocab.Vectors('../input/embeddings/glove.840B.300d/glove.840B.300d.txt')#, max_vectors=1000)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f35a7213fc9a7e80a7c210d11b3a8094d3a8e07e"},"cell_type":"code","source":"import operator \n\ndef check_coverage(vocab,embeddings_index):\n    a = {}\n    oov = {}\n    k = 0\n    i = 0\n    for word in tqdm(vocab):\n        try:\n            a[word] = embeddings_index[word]\n            k += vocab[word]\n        except:\n\n            oov[word] = vocab[word]\n            i += vocab[word]\n            pass\n\n    print('Found embeddings for {:.2%} of vocab'.format(len(a) / len(vocab)))\n    print('Found embeddings for  {:.2%} of all text'.format(k / (k + i)))\n    sorted_x = sorted(oov.items(), key=operator.itemgetter(1))[::-1]\n\n    return sorted_x","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"42c2b82740ac82f5678a46b7c82b5525616c1304","scrolled":false},"cell_type":"code","source":"oov = check_coverage(train_vocab, glove_vectors.stoi)","execution_count":null,"outputs":[]},{"metadata":{"_kg_hide-output":true,"trusted":true,"_uuid":"0e55005092e11441fb1c2df46f1f4192a4b12d68"},"cell_type":"code","source":"oov[:100]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c8da1f0acfc9ff52d9a482f52368a0fc78756ffa"},"cell_type":"code","source":"oov = check_coverage(test_vocab, glove_vectors.stoi)","execution_count":null,"outputs":[]},{"metadata":{"_kg_hide-output":true,"trusted":true,"scrolled":true,"_uuid":"653622f7233e2d8cbf81b4cbc1e30077f3fdf72b"},"cell_type":"code","source":"oov[:100]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e426f4f4fb8234ebf96586fe0eacca88f0c8837c"},"cell_type":"code","source":"del glove_vectors\ngc.collect()\n\nfor file in os.listdir('./.vector_cache/'):\n    os.remove(f'./.vector_cache/{file}')\n\nparagram_vectors = torchtext.vocab.Vectors('../input/embeddings/paragram_300_sl999/paragram_300_sl999.txt')#, max_vectors=1000)\n\noov = check_coverage(test_vocab, paragram_vectors.stoi)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"6e145f96b59d95afca91a8ccfb26ead1da149c4a"},"cell_type":"code","source":"oov = check_coverage(train_vocab, paragram_vectors.stoi)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"de0a78175815413d204d9b7f70b5facba97a7b58","_kg_hide-output":true},"cell_type":"code","source":"oov[:10]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"3e6f08d623ba4296c188c08111da89fda7d2851a"},"cell_type":"code","source":"oov = check_coverage(test_vocab, paragram_vectors.stoi)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"3299e879a9850595e983ee2d48f59be4d8106503","_kg_hide-output":true,"scrolled":true},"cell_type":"code","source":"oov[:10]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"932de03f6e2f084a3d504fe59625d8cbb0dd99ee"},"cell_type":"code","source":"del paragram_vectors\ngc.collect()\n\nfor file in os.listdir('./.vector_cache/'):\n    os.remove(f'./.vector_cache/{file}')\n\nfasttext_vectors = torchtext.vocab.Vectors('../input/embeddings/wiki-news-300d-1M/wiki-news-300d-1M.vec')#, max_vectors=1000)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"0ee7bdd4f8619296fb7728a2c972a0427af25d7e"},"cell_type":"code","source":"oov = check_coverage(train_vocab, fasttext_vectors.stoi)","execution_count":null,"outputs":[]},{"metadata":{"_kg_hide-output":true,"trusted":true,"_uuid":"41ea94f3d06c3c65b58ff3cececa811cc254f381"},"cell_type":"code","source":"oov[:100]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"8730e9b0da310cc8e4ed9cc5357cc36573a9ec15"},"cell_type":"code","source":"oov = check_coverage(test_vocab, fasttext_vectors.stoi)","execution_count":null,"outputs":[]},{"metadata":{"_kg_hide-output":true,"trusted":true,"_uuid":"c0577d3cb57a17a4a605748c4204efd6083071d1"},"cell_type":"code","source":"oov[:100]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d63ed5a15296d1fcbfb1fa50794cb47e71b09797"},"cell_type":"code","source":"for file in os.listdir('./.vector_cache/'):\n    os.remove(f'./.vector_cache/{file}')","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}