{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import numpy as np\nimport pandas as pd\n\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\nimport os\nimport gc\nimport re\nimport string\nimport time\n\n\nfrom pyphen import Pyphen\nfrom gensim.models import KeyedVectors as wv\nfrom sklearn.feature_extraction.text import TfidfVectorizer\n \n# from tqdm import tqdm_notebook as tqdm\nfrom tqdm import tqdm\nfrom collections import Counter\nfrom contextlib import contextmanager\nfrom functools import lru_cache\nfrom keras.preprocessing.text import text_to_word_sequence, Tokenizer\n\nimport Levenshtein as lv\n\npd.options.display.max_rows = 8\npd.options.display.max_columns = 999\nprint(os.listdir(\"../input\"))\n\n@contextmanager\ndef timer(name):\n    t0 = time.time()\n    yield\n    print(f'【{name}】 done in 【{time.time() - t0:.0f}】 s')","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"train_df = pd.read_csv('../input/train.csv')\ntest_df = pd.read_csv('../input/test.csv')\nsubmission = pd.read_csv('../input/sample_submission.csv')\nprint(train_df.shape, test_df.shape, submission.shape)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"76d6b1d90e39baed0acbab57110f325d5b492bc8"},"cell_type":"code","source":"s = test_df.qid\nprint(s.shape, s.nunique())\ns1 = submission.qid\nprint(s1.shape, s1.nunique())\nprint(np.sum(s!=s1))\ndel s,s1,submission\ngc.collect()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"6c9a0e06b0cbf189f269ecf2ac885188983d9a71"},"cell_type":"code","source":"y = train_df.target\nprint(y.shape, y.loc[1==y].shape, y.loc[1==y].shape[0]/y.shape[0])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e89ec851d14b360b498b68db1d86921cb3312fc0"},"cell_type":"code","source":"train_df.loc[0==y]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"299cda1da17dc7bbd058cbe39c3cad57cdabb35a"},"cell_type":"code","source":"train_df.loc[1==y]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e3acc2cb835776677a114ef6ce5ef47c899c5197"},"cell_type":"code","source":"s = train_df.loc[1==y, 'question_text'].str.len()\nprint(s.describe())\nprint()\ns = train_df.loc[0==y, 'question_text'].str.len()\nprint(s.describe())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"da07d568aea3593f19d01b7c801f67a5eaa5d977"},"cell_type":"code","source":"s = train_df.question_text.str.len()\nprint(s.describe())\nprint()\ns = test_df.question_text.str.len()\nprint(s.describe())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"4b4482b0a0ddec12066b7e77b659a78cc2145668"},"cell_type":"code","source":"s = train_df.loc[1==y, 'question_text'].str.count(' ')\nprint(s.describe())\nprint()\ns = train_df.loc[0==y, 'question_text'].str.count(' ')\nprint(s.describe())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"1dc6246a9706fc7764d446c568522786375ae28b"},"cell_type":"code","source":"s = train_df.question_text.str.count(' ')\nprint(s.describe())\nprint()\ns = test_df.question_text.str.count(' ')\nprint(s.describe())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"dc29c84d9678766ef5da716978a417a4ca4dce74"},"cell_type":"code","source":"s = train_df.loc[train_df.target==1,'question_text'].str.count(' ').sort_values()\nprint(s.quantile([0.9,0.95,0.99,0.995,0.999,0.9995,0.9999,0.9999,0.99995,0.99999,0.999995,0.999999]).to_dict())\nprint(s.iloc[-30:].to_dict())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d0fd5b2c5fadd31783d900de2ca2bf69c6f05208"},"cell_type":"code","source":"s = train_df.loc[train_df.target==0,'question_text'].str.count(' ').sort_values()\nprint(s.quantile([0.9,0.95,0.99,0.995,0.999,0.9995,0.9999,0.9999,0.99995,0.99999,0.999995,0.999999]).to_dict())\nprint(s.iloc[-30:].to_dict())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"0de35823c9bfd348ee7dcbf497dc2a2848bedd13"},"cell_type":"code","source":"s = train_df.question_text.str.count(' ').sort_values()\nprint(s.quantile([0.9,0.95,0.99,0.995,0.999,0.9995,0.9999,0.9999,0.99995,0.99999,0.999995,0.999999]).to_dict())\nprint(s.iloc[-30:].to_dict())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"cf3a267160f0afd7d811cf1423f46ca9200e4a28"},"cell_type":"code","source":"s = test_df.question_text.str.count(' ').sort_values()\nprint(s.quantile([0.9,0.95,0.99,0.995,0.999,0.9995,0.9999,0.9999,0.99995,0.99999,0.999995,0.999999]).to_dict())\nprint(s.iloc[-30:].to_dict())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e04e54b68a2f298c6f66bb826a40fe1691e30790"},"cell_type":"code","source":"for min_df in [1,2,3,4,5,10,20,30]:\n    tvr = TfidfVectorizer(token_pattern=r'(?u)\\w+|[^\\w\\s]', strip_accents='unicode', min_df=min_df)\n    gc.collect()\n    tvr.fit(train_df.question_text)\n    tr_words = tvr.get_feature_names()\n    tvr.fit(test_df.question_text)\n    ts_words = tvr.get_feature_names()\n    rs_words = np.setdiff1d(ts_words, tr_words)\n    tt_words1 = np.union1d(tr_words, ts_words)\n    tvr.fit(train_df.question_text.append(test_df.question_text))\n    tt_words2 = tvr.get_feature_names()\n    tvr.fit(train_df.loc[train_df.target>0, 'question_text'])\n    pos_words = tvr.get_feature_names()\n    tvr.fit(train_df.loc[0==train_df.target, 'question_text'])\n    neg_words = tvr.get_feature_names()\n    print(f'min_df={min_df}: tr_words={len(tr_words)}, ts_words={len(ts_words)}, rs_words={rs_words.shape}, '\n          +f'tt_words1={tt_words1.shape}, tt_words2={len(tt_words2)}, pos_words={len(pos_words)}, neg_words={len(neg_words)}')\n    del tvr,tr_words,ts_words,rs_words,tt_words1,tt_words2,pos_words,neg_words\n    gc.collect()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"980bc01a438e084f1e1f85208d955240956a601a"},"cell_type":"code","source":"@lru_cache(maxsize=65536)\ndef remove_punctuation(text):\n    return ''.join(ch for ch in text if ch not in string.punctuation)\n\n\n@lru_cache(maxsize=65536)\ndef count_syllable(word):\n    dic = Pyphen(lang='en_US')\n    word_hyphenated = dic.inserted(word)\n    return max(1, word_hyphenated.count(\"-\") + 1)\n\n\n@lru_cache(maxsize=65536)\ndef syllable_count(text, lang='en_US'):\n    text = text.lower()\n    text = remove_punctuation(text)\n\n    if not text:\n        return 0\n\n    dic = Pyphen(lang=lang)\n    count = 0\n    for word in text.split(' '):\n        count += count_syllable(word)\n    return count\n\n\n@lru_cache(maxsize=65536)\ndef lexicon_count(text, removepunct=True):\n    if removepunct:\n        text = remove_punctuation(text)\n    count = len(text.split())\n    return count\n\n\n@lru_cache(maxsize=65536)\ndef sentence_count(text):\n    ignore_count = 0\n    sentences = re.split(r' *[.?!][\\'\")\\]]*[ |\\n](?=[A-Z])', text)\n    for sentence in sentences:\n        if lexicon_count(sentence) <= 2:\n            ignore_count += 1\n    return max(1, len(sentences) - ignore_count)\n\n\n@lru_cache(maxsize=65536)\ndef polysyllabcount(text):\n    count = 0\n    for word in text.split():\n        wrds = syllable_count(word)\n        if wrds >= 3:\n            count += 1\n    return count\n\n\n@lru_cache(maxsize=65536)\ndef linsear_write_formula(text):\n    easy_word = 0\n    difficult_word = 0\n    text_list = text.split()[:100]\n\n    for word in text_list:\n        if syllable_count(word) < 3:\n            easy_word += 1\n        else:\n            difficult_word += 1\n\n    text = ' '.join(text_list)\n\n    number = easy_word * 1 + difficult_word * 3 / sentence_count(text)\n\n    if number <= 20:\n        number -= 2\n\n    return number / 2\n\n\n@lru_cache(maxsize=65536)\ndef lix(text, avg_sentence_length):\n    words = text.split()\n\n    words_len = len(words)\n    long_words = len([wrd for wrd in words if len(wrd) > 6])\n\n    per_long_words = long_words * 100 / words_len\n\n    return avg_sentence_length + per_long_words","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f6e170db9cecd0303e7c0b49d1640b12efe91ffa"},"cell_type":"code","source":"def encode_text(df):\n    def count_chars(txt):\n        _len = 0\n        digit_cnt, number_cnt = 0, 0\n        lower_cnt, upper_cnt, letter_cnt, word_cnt = 0, 0, 0, 0\n        char_cnt, term_cnt = 0, 0\n        conj_cnt, blank_cnt, punc_cnt = 0, 0, 0\n        sign_cnt, marks_cnt = 0, 0\n\n        flag = 10\n        for ch in txt:\n            _len += 1\n            if ch in string.ascii_lowercase:\n                lower_cnt += 1\n                letter_cnt += 1\n                char_cnt += 1\n                if flag:\n                    word_cnt += 1\n                    if flag > 2:\n                        term_cnt += 1\n                    flag = 0\n            elif ch in string.ascii_uppercase:\n                upper_cnt += 1\n                letter_cnt += 1\n                char_cnt += 1\n                if flag:\n                    word_cnt += 1\n                    if flag > 2:\n                        term_cnt += 1\n                    flag = 0\n            elif ch in string.digits:\n                digit_cnt += 1\n                char_cnt += 1\n                if 1 != flag:\n                    number_cnt += 1\n                    if flag > 2:\n                        term_cnt += 1\n                    flag = 1\n            elif '_' == ch:\n                conj_cnt += 1\n                char_cnt += 1\n                if flag > 2:\n                    term_cnt += 1\n                flag = 2\n            elif ch in string.whitespace:\n                blank_cnt += 1\n                flag = 3\n            elif ch in string.punctuation:\n                punc_cnt += 1\n                flag = 4\n            else:\n                sign_cnt += 1\n                if flag != 5:\n                    marks_cnt += 1\n                    flag = 5\n\n        syllable_cnt = syllable_count(txt)\n        sentence_cnt = sentence_count(txt)\n        avg_sentence_length = word_cnt / sentence_cnt\n        avg_sentence_per_word = sentence_cnt / max(1, word_cnt)\n        avg_syllables_per_word = syllable_cnt / max(1, word_cnt)\n        avg_character_per_word = char_cnt / max(1, word_cnt)\n        avg_letter_per_word = letter_cnt / max(1, word_cnt)\n        flesch_reading_ease = 206.835 - 1.015 * avg_sentence_length - 84.6 * avg_syllables_per_word\n        flesch_kincaid_grade = 0.39 * avg_sentence_length + 11.8 * avg_syllables_per_word - 15.59\n        polysyllable_cnt = polysyllabcount(txt)\n        smog_index = (1.043 * (30 * polysyllable_cnt / sentence_cnt) ** .5) + 3.1291\n        coleman_liau_index = 5.8 * avg_letter_per_word - 29.6 * avg_sentence_per_word - 15.8\n        readability = 4.71 * avg_character_per_word + 0.5 * avg_sentence_length - 21.43\n        linsear_write_metric = linsear_write_formula(txt)\n        lix_metric = lix(txt, avg_sentence_length)\n\n        return (_len, digit_cnt, number_cnt, digit_cnt / max(1, number_cnt), lower_cnt, upper_cnt, letter_cnt,\n                word_cnt, avg_letter_per_word, char_cnt, term_cnt, char_cnt / max(1, term_cnt), conj_cnt,\n                blank_cnt, punc_cnt, sign_cnt, marks_cnt, sign_cnt / max(1, marks_cnt), syllable_cnt,\n                sentence_cnt, avg_sentence_length, avg_sentence_per_word, avg_syllables_per_word,\n                avg_character_per_word, flesch_reading_ease, flesch_kincaid_grade, polysyllable_cnt, smog_index,\n                coleman_liau_index, readability, linsear_write_metric, lix_metric)\n\n    (df['char_len'], df['digit_cnt'], df['number_cnt'], df['digit_cnt/number_cnt'], df['lower_cnt'], df['upper_cnt'],\n     df['letter_cnt'], df['word_cnt'], df['avg_letter_per_word'], df['char_cnt'], df['term_cnt'], df['char_cnt/term_cnt'],\n     df['conj_cnt'], df['blank_cnt'], df['punc_cnt'], df['sign_cnt'], df['marks_cnt'], df['sign_cnt/marks_cnt'], df['syllable_cnt'], \n     df['sentence_cnt'], df['avg_sentence_length'], df['avg_sentence_per_word'], df['avg_syllables_per_word'], \n     df['avg_character_per_word'],df['flesch_reading_ease'],df['flesch_kincaid_grade'],df['polysyllable_cnt'],df['smog_index'],\n     df['coleman_liau_index'],df['readability'],df['linsear_write_metric'],df['lix_metric']) = zip(\n        *df.question_text.apply(count_chars))\n\n    return df","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"8bc071e8aab4dc60cddc02d6d77dcb068ccbc378"},"cell_type":"code","source":"with timer('encode train text'):\n    train_df = encode_text(train_df)\ntrain_df","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d28803489abf621d0dfbae1295e4fc5dbe24e130"},"cell_type":"code","source":"test_df = encode_text(test_df)\ntest_df","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a3a6555b0a85ab8aedae5bc9acfe5b600f5b4a55"},"cell_type":"code","source":"train_df.info()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"4110d5fa245a594029770040135f0ff07334e358"},"cell_type":"code","source":"test_df.info()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"7c324faaaedae5ad2cfdb8537e0f4619fc3786d3"},"cell_type":"markdown","source":"# embed"},{"metadata":{"trusted":true,"_uuid":"6e3ee8cce53db6b6bf50388bfa1f8e17f091b00d"},"cell_type":"code","source":"train_df = train_df.fillna('the')\ntest_df = test_df.fillna('the')\ngc.collect()\n\nwith timer('reserve punctuation'):\n    def reserve_puncts(text):\n        puncts = [',', '.', '\"', ':', ')', '(', '-', '!', '?', '|', ';', \"'\", '$', '&', '/', '[', ']', '>', '%',\n                  '=', '#', '*', '+', '\\\\', '•', '~', '@', '£', '·', '_', '{', '}', '©', '^', '®', '`', '<', '→',\n                  '°', '€', '™', '›', '♥', '←', '×', '§', '″', '′', 'Â', '█', '½', 'à', '…', '“', '★', '”', '–',\n                  '●', 'â', '►', '−', '¢', '²', '¬', '░', '¶', '↑', '±', '¿', '▾', '═', '¦', '║', '―', '¥', '▓',\n                  '—', '‹', '─', '▒', '：', '¼', '⊕', '▼', '▪', '†', '■', '’', '▀', '¨', '▄', '♫', '☆', 'é', '¯',\n                  '♦', '¤', '▲', 'è', '¸', '¾', 'Ã', '⋅', '‘', '∞', '∙', '）', '↓', '、', '│', '（', '»', '，', '♪',\n                  '╩', '╚', '³', '・', '╦', '╣', '╔', '╗', '▬', '❤', 'ï', 'Ø', '¹', '≤', '‡', '√']\n        punct_dic = {punct: f' {punct} ' for punct in puncts}\n        punct_dic.update({'\\t': ' ', '\\n': ' ', '\\r': ' ', '\\u200b': ''})\n        return text.translate(str.maketrans(punct_dic))\n\n    train_df['question_text'] = train_df.question_text.apply(reserve_puncts)\n    test_df['question_text'] = test_df.question_text.apply(reserve_puncts)\n    gc.collect()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b28bc0af49f692112a645e7e706817129c54d3ba"},"cell_type":"code","source":"tr_tkr = Tokenizer(filters='')\ntr_tkr.fit_on_texts(train_df.question_text)\nts_tkr = Tokenizer(filters='')\nts_tkr.fit_on_texts(test_df.question_text)\ntt_tkr = Tokenizer(filters='')\ntt_tkr.fit_on_texts(train_df.question_text.append(test_df.question_text))\npos_tkr = Tokenizer(filters='')\npos_tkr.fit_on_texts(train_df.loc[train_df.target==1,'question_text'])\nneg_tkr = Tokenizer(filters='')\nneg_tkr.fit_on_texts(train_df.loc[train_df.target==0,'question_text'])\ngc.collect()\n\nfor min_df in [1,2,3,4,5,10,20,30]:\n    tr_words = [word for word,cnt in tr_tkr.word_counts.items() if cnt>=min_df]\n    ts_words = [word for word,cnt in ts_tkr.word_counts.items() if cnt>=min_df]\n    rs_words = np.setdiff1d(ts_words, tr_words)\n    tt_words1 = np.union1d(tr_words, ts_words)\n    tt_words2 = [word for word,cnt in tt_tkr.word_counts.items() if cnt>=min_df]\n    pos_words = [word for word,cnt in pos_tkr.word_counts.items() if cnt>=min_df]\n    neg_words = [word for word,cnt in neg_tkr.word_counts.items() if cnt>=min_df]\n    print(f'min_df={min_df}: tr_words={len(tr_words)}, ts_words={len(ts_words)}, rs_words={rs_words.shape}, '\n          +f'tt_words1={tt_words1.shape}, tt_words2={len(tt_words2)}, pos_words={len(pos_words)}, neg_words={len(neg_words)}')\n    del tr_words,ts_words,rs_words,tt_words1,tt_words2,pos_words,neg_words\n    gc.collect()\n\ntr_words = list(tr_tkr.word_index.keys())\nts_words = list(ts_tkr.word_index.keys())\nrs_words = np.setdiff1d(ts_words, tr_words)\ntt_words = list(tt_tkr.word_index.keys())\npos_words = list(pos_tkr.word_index.keys())\nneg_words = list(neg_tkr.word_index.keys())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"971ba5b69021aa11560b96727d694a9a513998b5"},"cell_type":"code","source":"def load_embed_dic(embed_id, embed_root_dir='../input/embeddings'):\n    def get_coefs(word, *arr):\n        return word, np.asarray(arr, dtype='float32')\n\n    file_path_dic = {\n        'glove': os.path.join(embed_root_dir, 'glove.840B.300d', 'glove.840B.300d.txt'),\n        'wiki': os.path.join(embed_root_dir, 'wiki-news-300d-1M', 'wiki-news-300d-1M.vec'),\n        'para': os.path.join(embed_root_dir, 'paragram_300_sl999', 'paragram_300_sl999.txt'),\n        'google': os.path.join(embed_root_dir, 'GoogleNews-vectors-negative300', 'GoogleNews-vectors-negative300.bin')\n    }\n    if 'wiki' == embed_id:\n        embed_dic = dict(get_coefs(*line.split(' ')) for line in open(\n            file_path_dic[embed_id], encoding='utf8', errors='ignore') if len(line) > 100)\n    elif 'google' == embed_id:\n        embed_dic = wv.load_word2vec_format(file_path_dic[embed_id], binary=True)\n    else:\n        embed_dic = dict(get_coefs(*line.split(' ')) for line in open(\n            file_path_dic[embed_id], encoding='utf8', errors='ignore'))\n\n    return embed_dic\n\n\ndef set_diff(s1,s2,batch_num=100):\n    batch_size = len(s2) // batch_num + 1\n    for i in range(batch_num):\n        s = s2[i*batch_size: (i+1)*batch_size]\n        s1 = np.setdiff1d(s1, s)\n        del s\n        gc.collect()\n    return s1\n\n\n@lru_cache(maxsize=65536)\ndef word_distance(_word1, _word2):\n    return lv.distance(_word1, _word2)\n\n\ndef similar(_word1, _word2):\n    _len1 = len(_word1)\n    _len2 = len(_word2)\n    _len = min(_len1, _len2)\n    if _len<5 or (_len>=5 and abs(_len1-_len2)>1):\n        return False\n    return word_distance(_word1, _word2)<=1\n\n\n_digits = re.compile('\\d')\n@lru_cache(maxsize=65536)\ndef contains_digits(d):\n    return bool(_digits.search(d))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e6edbf8257eb7eee63bb869a88262e9080f77d5f"},"cell_type":"code","source":"embed_dic = load_embed_dic('glove')\nglove_words = list(embed_dic.keys())\ndel embed_dic\ngc.collect()\nprint(f'glove_words: {len(glove_words)}')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"fed52f295faefb7699e7ea16faafc335fc7543bb"},"cell_type":"code","source":"with timer('train lv'):\n    tr_miss_words = set_diff(tr_words, glove_words)\n    tr_embed_words = np.setdiff1d(tr_words, tr_miss_words)\n    len_dic = {}\n    for word in tr_embed_words:\n        _len = len(word)\n        if _len in len_dic:\n            len_dic[_len].append(word)\n        else:\n            len_dic[_len] = [word]\n            \n    tr_miss_word_dic = {}\n    for word1 in tr_miss_words:\n        if not contains_digits(word1):\n            _len = len(word1)\n            _embed_words = []\n            if _len-1 in len_dic:\n                _embed_words += len_dic[_len-1]\n            if _len in len_dic:\n                _embed_words += len_dic[_len]\n            if _len+1 in len_dic:\n                _embed_words += len_dic[_len+1]\n\n            cand_word = None\n            for word2 in _embed_words:\n                if similar(word1, word2):\n                    if word2>=word1:\n                        tr_miss_word_dic[word1] = word2\n                        break\n                    else:\n                        cand_word = word2\n            if word1 not in tr_miss_word_dic and cand_word is not None:\n                tr_miss_word_dic[word1] = cand_word\n    print(f'tr_miss_words: {len(tr_miss_words)}, tr_embed_words: {tr_embed_words.shape}, tr_miss_word_dic: {len(tr_miss_word_dic)}')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"4df677b6909bbfcc9cd19ad5618249e5c5ed2666"},"cell_type":"code","source":"print(tr_miss_word_dic) ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b932c6246383060cf66c185b6afe2a6e2669ca62"},"cell_type":"code","source":"with timer('test lv'):\n    ts_miss_words = set_diff(ts_words, glove_words)\n    print(f'before, ts_miss_words: {ts_miss_words.shape}', end='; ')\n    ts_miss_words = np.setdiff1d(ts_miss_words, list(tr_miss_word_dic.keys()))\n    print(f'after, ts_miss_words: {ts_miss_words.shape}')\n    len_dic = {}\n    for word in tr_embed_words:\n        _len = len(word)\n        if _len in len_dic:\n            len_dic[_len].append(word)\n        else:\n            len_dic[_len] = [word]\n            \n    ts_miss_word_dic = {}\n    for word1 in tqdm(ts_miss_words):\n        if not contains_digits(word1):\n            _len = len(word1)\n            _embed_words = []\n            if _len-1 in len_dic:\n                _embed_words += len_dic[_len-1]\n            if _len in len_dic:\n                _embed_words += len_dic[_len]\n            if _len+1 in len_dic:\n                _embed_words += len_dic[_len+1]\n\n            cand_word = None\n            for word2 in _embed_words:\n                if similar(word1, word2):\n                    if word2>=word1:\n                        ts_miss_word_dic[word1] = word2\n                        break\n                    else:\n                        cand_word = word2\n            if word1 not in ts_miss_word_dic and cand_word is not None:\n                ts_miss_word_dic[word1] = cand_word\n    print(f'ts_miss_word_dic: {len(ts_miss_word_dic)}')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"bceae79a6dbe5282e1a866b7659d6468ae956867"},"cell_type":"code","source":"print(ts_miss_word_dic)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e67662b6754d0c70a2a3630fbe818bc5faee7d39"},"cell_type":"code","source":"del tr_miss_words,tr_embed_words,len_dic,tr_miss_word_dic,ts_miss_words,ts_miss_word_dic\ngc.collect()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"da340884d2e3a7527c201a5064d96bfaaad91d90"},"cell_type":"code","source":"embed_dic = load_embed_dic('glove')\nglove_words = list(embed_dic.keys())\nwvs = np.stack(embed_dic.values())\nprint(wvs.shape)\ndel embed_dic\ngc.collect()\n\nwvs = np.sort(np.ravel(wvs))\ngc.collect()\nprint(wvs.shape, np.mean(wvs), np.std(wvs))\n\ny = wvs[::100]\ndel wvs\ngc.collect()\nfig = plt.figure(figsize=(18, 9))\nsns.distplot(y)\ndel y\ngc.collect()\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e5756024469ae845645b44e065b5d64a8e5b2a20"},"cell_type":"code","source":"embed_dic = load_embed_dic('wiki')\nwiki_words = list(embed_dic.keys())\nwvs = np.stack(embed_dic.values())\nprint(wvs.shape)\ndel embed_dic\ngc.collect()\n\nwvs = np.sort(np.clip(np.ravel(wvs),-5,5))\ngc.collect()\nprint(wvs.shape, np.mean(wvs), np.std(wvs))\n\ny = wvs[::100]\ndel wvs\ngc.collect()\nfig = plt.figure(figsize=(18, 9))\nsns.distplot(y)\ndel y\ngc.collect()\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"4d519ec1f9b3970e61ee94f5a9ac553ac13970cf"},"cell_type":"code","source":"embed_dic = load_embed_dic('para')\npara_words = list(embed_dic.keys())\nwvs = np.stack(embed_dic.values())\nprint(wvs.shape)\ndel embed_dic\ngc.collect()\n\nwvs = np.sort(np.ravel(wvs))\ngc.collect()\nprint(wvs.shape, np.mean(wvs), np.std(wvs))\n\ny = wvs[::100]\ndel wvs\ngc.collect()\nfig = plt.figure(figsize=(18, 9))\nsns.distplot(y)\ndel y\ngc.collect()\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"0a2a961be20d3a161a7d4c2fe19344b18f8eefe3"},"cell_type":"code","source":"embed_dic = load_embed_dic('google')\ngoogle_words = list(embed_dic.vocab.keys())\nwvs = embed_dic.vectors\nprint(wvs.shape)\ndel embed_dic\ngc.collect()\n\nwvs = np.sort(np.ravel(wvs))\ngc.collect()\nprint(wvs.shape, np.mean(wvs), np.std(wvs))\n\ny = wvs[::100]\ndel wvs\ngc.collect()\nfig = plt.figure(figsize=(18, 9))\nsns.distplot(y)\ndel y\ngc.collect()\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e7f691d5d8418d560991f0163bb76eb9dde45eef"},"cell_type":"code","source":"wv_names = ['glove','para','wiki','google']\nwv_words = [glove_words,para_words,wiki_words,google_words]\ndata_names = ['tr','ts','rs','pos','neg']\ndata_words = [tr_words,ts_words,rs_words,pos_words,neg_words]\nmiss_words = [[set_diff(data_word, wv_word) for data_word in data_words] for wv_word in wv_words]\nprint(len(miss_words), len(miss_words[0]))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"2933348a29bcd0f4b79301f173d312e88b190a51"},"cell_type":"code","source":"def print_embed_info(embed_ids):\n    for i in range(len(embed_ids)):\n        for j in range(len(embed_ids[i])-1):\n            print(f'{wv_names[embed_ids[i][j]]} &', end=' ')\n        print(f'{wv_names[embed_ids[i][-1]]}:', end=' ')\n        \n        for j in range(len(data_names)-1):\n            cur_miss_words = miss_words[embed_ids[i][0]][j]\n            for k in range(1, len(embed_ids[i])):\n                cur_miss_words = np.intersect1d(cur_miss_words, miss_words[embed_ids[i][k]][j])\n            print(f'{data_names[j]}({len(cur_miss_words)}),', end=' ')\n        cur_miss_words = miss_words[embed_ids[i][0]][-1]\n        for k in range(1, len(embed_ids[i])):\n            cur_miss_words = np.intersect1d(cur_miss_words, miss_words[embed_ids[i][k]][-1])\n        print(f'{data_names[-1]}({len(cur_miss_words)})')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"0f27ae7eeae3e6c1b5a7fedc962e4a034d9bcafd"},"cell_type":"code","source":"print_embed_info([[0],[1],[2],[3]])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"1586ae0d0233fe8227d8932985fc6e1d2ff8eb18"},"cell_type":"code","source":"print_embed_info([[0,1],[0,2],[0,3],[1,2],[1,3],[2,3]])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ff7ee1129da898daee51ead3f66ced54b8ab9aa2"},"cell_type":"code","source":"print_embed_info([[0,1,2],[0,1,3],[0,2,3],[1,2,3]])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f2327944d80883346e5793561d4685924ddb719b"},"cell_type":"code","source":"print_embed_info([[0,1,2,3]])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"849592e565feebdd0511eebb0cbed086eb8ecdde"},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"}},"nbformat":4,"nbformat_minor":1}