{"cells":[{"metadata":{"_uuid":"84268fee90a2d1efb65816228ce1d1fbb2a88630"},"cell_type":"markdown","source":"# This kernel mainly focus on text prepocessing and embeddings. Three embedding models has been taken into consideration.\n- Deal with upper and lower case\n- Deal with contractions\n- Deal with special characters and  punctuations\n- Deal with most frequent mispellings\n- Deal with numbers\n\n# It shows that paragram embedding got the best coverage rate for this data.\n- Found embeddings for 74.31% of vocab\n- Found embeddings for  99.66% of all text"},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nimport os\nprint(os.listdir(\"../input\"))\n\n# Any results you write to the current directory are saved as output.","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"import nltk\nimport pandas as pd\nimport numpy as np\nimport operator \nimport re\nfrom tqdm import tqdm\ntqdm.pandas()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"6407502abd40038e56d7ecad24ef4f71fb5712a5"},"cell_type":"markdown","source":"Experiment keras tokenizer. Learning from RNN+4"},{"metadata":{"trusted":true,"_uuid":"e0105c0bb49716675d257b1a3aa3565bdd93df78"},"cell_type":"code","source":"from keras.preprocessing.text import Tokenizer","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a29274a07964a10026880be703da76654b75ddf3"},"cell_type":"code","source":"train = pd.read_csv(\"../input/train.csv\")\ntest = pd.read_csv(\"../input/test.csv\")\ndf = pd.concat([train ,test])\n\nprint(\"Number of texts: \", df.shape[0])","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"8d6d839ef7b0166b9e9c45f7f600038d279293ab"},"cell_type":"markdown","source":"> # Load embeddings"},{"metadata":{"trusted":true,"_uuid":"36b833b4d77d033bd64e3d941e0d92f72c40ed11"},"cell_type":"code","source":"def load_embed(file):\n    def get_coefs(word,*arr): \n        return word, np.asarray(arr, dtype='float32')\n    if file == '../input/embeddings/wiki-news-300d-1M/wiki-news-300d-1M.vec':\n        embeddings_index = dict(get_coefs(*o.split(\" \")) for o in open(file) if len(o)>100)\n    else:\n        embeddings_index = dict(get_coefs(*o.split(\" \")) for o in open(file, encoding='latin'))\n    return embeddings_index","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"scrolled":true,"_uuid":"62ab34d5beb1dbcfecbfdbfa1b3263351115c07a"},"cell_type":"code","source":"## Three different embeddings\n# glove = '../input/embeddings/glove.840B.300d/glove.840B.300d.txt'\nparagram =  '../input/embeddings/paragram_300_sl999/paragram_300_sl999.txt'\n# wiki_news = '../input/embeddings/wiki-news-300d-1M/wiki-news-300d-1M.vec'\n\n# print(\"Extracting GloVe embedding\")\n# embed_glove = load_embed(glove)\nprint(\"Extracting Paragram embedding\")\nembed_paragram = load_embed(paragram)\n# print(\"Extracting FastText embedding\")\n# embed_fasttext = load_embed(wiki_news)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"c13bec8ae16d73fae3f113d16c9edacd7a6ec14f"},"cell_type":"markdown","source":"# Convert train data to words"},{"metadata":{"trusted":true,"_uuid":"a6cf9ce56f3ed131cd8bd3be55c2eb282ecae35f"},"cell_type":"code","source":"# function to convert text data to vocabulary (also can use python default dictionary and nltk.dict)\ndef build_vocab(texts):\n    sentences = texts.apply(lambda x: x.split()).values\n    vocab = {}\n    for sentence in sentences:\n        for word in sentence:\n            try:\n                vocab[word] += 1\n            except KeyError:\n                vocab[word] = 1\n    return vocab","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"688504f497999bfb02debdfb4e26b41371f86c01"},"cell_type":"code","source":"# vocab = build_vocab(df['question_text'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"scrolled":true,"_uuid":"14ebdbd90e0624c7396db43f737d354dfe45eaf5"},"cell_type":"code","source":"# Print first 5 words and their frequency\n# print({k: vocab[k] for k in list(vocab)[:5]})","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"9913b27ee485e142fccc0b350a84d6cb84a2824e"},"cell_type":"markdown","source":"# Build function to check embedding method's coverage rate"},{"metadata":{"trusted":true,"_uuid":"31ed46a488552f58995e2f4234e958e7b5bdfdd8"},"cell_type":"code","source":"# function to check embedding method's coverage rate\ndef check_coverage(vocab, embeddings_index):\n    known_words = {}\n    unknown_words = {}\n    nb_known_words = 0\n    nb_unknown_words = 0\n    for word in vocab.keys():\n        try:\n            known_words[word] = embeddings_index[word]\n            nb_known_words += vocab[word]\n        except:\n            unknown_words[word] = vocab[word]\n            nb_unknown_words += vocab[word]\n            pass\n\n    print('Found embeddings for {:.2%} of vocab'.format(len(known_words) / len(vocab)))\n    print('Found embeddings for  {:.2%} of all text'.format(nb_known_words / (nb_known_words + nb_unknown_words)))\n    unknown_words = sorted(unknown_words.items(), key=operator.itemgetter(1))[::-1]\n\n    return unknown_words","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"337aaaf31cab8d918a6e445506c372fc9954a8ab"},"cell_type":"markdown","source":"# (1)Deal with upper and lower case"},{"metadata":{"trusted":true,"_uuid":"ddf41ff0c4da3f9ead084999a06064b55692efa8"},"cell_type":"code","source":"# Lowercase text data\n    ## Build lowercase vocabulary for text data\ndf['lowered_question'] = df['question_text'].apply(lambda x: x.lower())\n# vocab_low = build_vocab(df['lowered_question'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f80def42f08d8fde2abccd7f124abf45590bcf91"},"cell_type":"code","source":"# Add lower case words'embedding  \ndef add_lower(embedding, vocab):\n    count = 0\n    for word in vocab:\n        if word in embedding and word.lower() not in embedding:  \n            embedding[word.lower()] = embedding[word]\n            count += 1\n    print(f\"Added {count} words to embedding\")\n\n# print(\"Glove : \")\n# add_lower(embed_glove, vocab)\nprint(\"Paragram : \")\n# add_lower(embed_paragram, vocab)\n# print(\"FastText : \")\n# add_lower(embed_fasttext, vocab)\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"scrolled":true,"_uuid":"02634a201128600b5d7108fd228228ddf22d5919"},"cell_type":"code","source":"# print(\"Glove : \")\n# oov_glove = check_coverage(vocab_low, embed_glove)\nprint(\"Paragram : \")\n# oov_paragram = check_coverage(vocab_low, embed_paragram)\n# print(\"FastText : \")\n# oov_fasttext = check_coverage(vocab_low, embed_fasttext)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"scrolled":false,"_uuid":"cd2809b99d7e55427d04216da2cea8644c7a24ff"},"cell_type":"code","source":"# oov_glove[:50]","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"340358cc84c29131ef83b5960b91e63a847e5a61"},"cell_type":"markdown","source":"\n### It shows that there are \n  - (1) contraction problem \n  - (2) punctuation problem\n### we need to handle them one by one to improve coverage rate"},{"metadata":{"_uuid":"9537056c5fa379c87987e65ea73129f398db57ef"},"cell_type":"markdown","source":"# (2)Handle contractions"},{"metadata":{"trusted":true,"_uuid":"811f978fcb42068f4ff9b331f1d6be6263936522"},"cell_type":"code","source":"# build a contraction mapping dictionary \ncontraction_mapping = {\"ain't\": \"is not\", \"aren't\": \"are not\",\"can't\": \"cannot\", \"'cause\": \"because\", \"could've\": \"could have\", \"couldn't\": \"could not\", \"didn't\": \"did not\",  \"doesn't\": \"does not\", \"don't\": \"do not\", \"hadn't\": \"had not\", \"hasn't\": \"has not\", \"haven't\": \"have not\", \"he'd\": \"he would\",\"he'll\": \"he will\", \"he's\": \"he is\", \"how'd\": \"how did\", \"how'd'y\": \"how do you\", \"how'll\": \"how will\", \"how's\": \"how is\",  \"I'd\": \"I would\", \"I'd've\": \"I would have\", \"I'll\": \"I will\", \"I'll've\": \"I will have\",\"I'm\": \"I am\", \"I've\": \"I have\", \"i'd\": \"i would\", \"i'd've\": \"i would have\", \"i'll\": \"i will\",  \"i'll've\": \"i will have\",\"i'm\": \"i am\", \"i've\": \"i have\", \"isn't\": \"is not\", \"it'd\": \"it would\", \"it'd've\": \"it would have\", \"it'll\": \"it will\", \"it'll've\": \"it will have\",\"it's\": \"it is\", \"let's\": \"let us\", \"ma'am\": \"madam\", \"mayn't\": \"may not\", \"might've\": \"might have\",\"mightn't\": \"might not\",\"mightn't've\": \"might not have\", \"must've\": \"must have\", \"mustn't\": \"must not\", \"mustn't've\": \"must not have\", \"needn't\": \"need not\", \"needn't've\": \"need not have\",\"o'clock\": \"of the clock\", \"oughtn't\": \"ought not\", \"oughtn't've\": \"ought not have\", \"shan't\": \"shall not\", \"sha'n't\": \"shall not\", \"shan't've\": \"shall not have\", \"she'd\": \"she would\", \"she'd've\": \"she would have\", \"she'll\": \"she will\", \"she'll've\": \"she will have\", \"she's\": \"she is\", \"should've\": \"should have\", \"shouldn't\": \"should not\", \"shouldn't've\": \"should not have\", \"so've\": \"so have\",\"so's\": \"so as\", \"this's\": \"this is\",\"that'd\": \"that would\", \"that'd've\": \"that would have\", \"that's\": \"that is\", \"there'd\": \"there would\", \"there'd've\": \"there would have\", \"there's\": \"there is\", \"here's\": \"here is\",\"they'd\": \"they would\", \"they'd've\": \"they would have\", \"they'll\": \"they will\", \"they'll've\": \"they will have\", \"they're\": \"they are\", \"they've\": \"they have\", \"to've\": \"to have\", \"wasn't\": \"was not\", \"we'd\": \"we would\", \"we'd've\": \"we would have\", \"we'll\": \"we will\", \"we'll've\": \"we will have\", \"we're\": \"we are\", \"we've\": \"we have\", \"weren't\": \"were not\", \"what'll\": \"what will\", \"what'll've\": \"what will have\", \"what're\": \"what are\",  \"what's\": \"what is\", \"what've\": \"what have\", \"when's\": \"when is\", \"when've\": \"when have\", \"where'd\": \"where did\", \"where's\": \"where is\", \"where've\": \"where have\", \"who'll\": \"who will\", \"who'll've\": \"who will have\", \"who's\": \"who is\", \"who've\": \"who have\", \"why's\": \"why is\", \"why've\": \"why have\", \"will've\": \"will have\", \"won't\": \"will not\", \"won't've\": \"will not have\", \"would've\": \"would have\", \"wouldn't\": \"would not\", \"wouldn't've\": \"would not have\", \"y'all\": \"you all\", \"y'all'd\": \"you all would\",\"y'all'd've\": \"you all would have\",\"y'all're\": \"you all are\",\"y'all've\": \"you all have\",\"you'd\": \"you would\", \"you'd've\": \"you would have\", \"you'll\": \"you will\", \"you'll've\": \"you will have\", \"you're\": \"you are\", \"you've\": \"you have\" }","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"scrolled":true,"_uuid":"be306343fac212ea11589ff9e06fb056e578fdca"},"cell_type":"code","source":"# see if and which contraction(in the dictionary we built before) exsists in each embedding \ndef known_contractions(embed):\n    known = []\n    for contract in contraction_mapping:\n        if contract in embed:\n            known.append(contract)\n    return known\n\n# print(\"- Known Contractions -\")\n# print(\"   Glove :\")\n# print(known_contractions(embed_glove))\n# print(\"   Paragram :\")\n# print(known_contractions(embed_paragram))\n# print(\"   FastText :\")\n# print(known_contractions(embed_fasttext))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b0f4804841cd0b99ca82a6a9ede4a72ad4bbf167"},"cell_type":"code","source":"# clean contractions in text data before embedding \ndef clean_contractions(text, mapping):\n    specials = [\"’\", \"‘\", \"´\", \"`\"]\n    for s in specials:\n        text = text.replace(s, \"'\")\n    text = ' '.join([mapping[t] if t in mapping else t for t in text.split(\" \")])\n    return text\ndf['treated_question'] = df['lowered_question'].apply(lambda x: clean_contractions(x, contraction_mapping))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"673c1e60f9e3197215acc11cee8b8526c4184967"},"cell_type":"code","source":"#Build vocabulary of text data after cleaning contractions\n    # check coverage of each embedding methods\n# vocab = build_vocab(df['treated_question'])\n# print(\"Glove : \")\n# oov_glove = check_coverage(vocab, embed_glove)\nprint(\"Paragram : \")\n# oov_paragram = check_coverage(vocab, embed_paragram)\n# print(\"FastText : \")\n# oov_fasttext = check_coverage(vocab, embed_fasttext)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"0ae2ff1197fc72b75dc877d69cb162adf296ac5e"},"cell_type":"markdown","source":"# (3)Deal with spectial characters"},{"metadata":{"trusted":true,"_uuid":"f487e8183c0fbb43e39dadeb88193ed09b2e28a6"},"cell_type":"code","source":"# Build a list of spectial characters\npunct = \"/-'?!.,#$%\\'()*+-/:;<=>@[\\\\]^_`{|}~\" + '\"\"“”’' + '∞θ÷α•à−β∅³π‘₹´°£€\\×™√²—–&'","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"6aef7d3917a4f2c428a732fb6ca429df4bfd6830"},"cell_type":"code","source":"# see if and which spectial characters(in the list we built before) exsists in each embedding \ndef unknown_punct(embed, punct):\n    unknown = ''\n    for p in punct:\n        if p not in embed:\n            unknown += p\n            unknown += ' '\n    return unknown\n\n# print(\"Glove :\")\n# print(unknown_punct(embed_glove, punct))\nprint(\"Paragram :\")\n# print(unknown_punct(embed_paragram, punct))\n# print(\"FastText :\")\n# print(unknown_punct(embed_fasttext, punct))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"517171e7cc6f027bf4e8ca520c29dc426b62a2ed"},"cell_type":"code","source":"punct_mapping = {\"‘\": \"'\", \"₹\": \"e\", \"´\": \"'\", \"°\": \"\", \"€\": \"e\", \"™\": \"tm\", \"√\": \" sqrt \", \"×\": \"x\", \"²\": \"2\", \"—\": \"-\", \"–\": \"-\", \"’\": \"'\", \"_\": \"-\", \"`\": \"'\", '“': '\"', '”': '\"', '“': '\"', \"£\": \"e\", '∞': 'infinity', 'θ': 'theta', '÷': '/', 'α': 'alpha', '•': '.', 'à': 'a', '−': '-', 'β': 'beta', '∅': '', '³': '3', 'π': 'pi','\\u200b': ' ', '…': ' ... ', '\\ufeff': '', 'करना': '', 'है': '' }","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ac0d528fc970e58cba88dafb25811f0e8935a0bd"},"cell_type":"code","source":"# clean special characters in text data before embedding \ndef clean_special_chars(text, punct, mapping):\n    \n    ## use a map to replace unknown characters with known ones.\n    for p in mapping:\n        text = text.replace(p, mapping[p])\n    ## make sure there are spaces between words and punctuation\n    for p in punct:\n        text = text.replace(p, f' {p} ')\n        \n    return text\n\ndf['treated_question'] = df['treated_question'].apply(lambda x: clean_special_chars(x, punct, punct_mapping))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"53923648f4e03c92df7924cb0357ea149d820364"},"cell_type":"code","source":"# # Build vocabulary of text data after cleaning special characters\n    ## check coverage of each embedding methods\n# vocab = build_vocab(df['treated_question'])\n# print(\"Glove : \")\n# oov_glove = check_coverage(vocab, embed_glove)\nprint(\"Paragram : \")\n# oov_paragram = check_coverage(vocab, embed_paragram)\n# print(\"FastText : \")\n# oov_fasttext = check_coverage(vocab, embed_fasttext)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"1ff2d8d0a87cd4c099a80452cd175f9ebca42237"},"cell_type":"markdown","source":"# (4)Deal with mispells (only can manually correct most frequent mispells)"},{"metadata":{"trusted":true,"_uuid":"e43b56839993e1b6ef4afc3607fbc2696599e114"},"cell_type":"code","source":"# build a word mapping dictionary for frequent mispells \nmispell_dict = {'colour': 'color', 'centre': 'center', 'favourite': 'favorite', 'travelling': 'traveling', 'counselling': 'counseling', 'theatre': 'theater', 'cancelled': 'canceled', 'labour': 'labor', 'organisation': 'organization', 'wwii': 'world war 2', 'citicise': 'criticize', 'youtu ': 'youtube ', 'Qoura': 'Quora', 'sallary': 'salary', 'Whta': 'What', 'narcisist': 'narcissist', 'howdo': 'how do', 'whatare': 'what are', 'howcan': 'how can', 'howmuch': 'how much', 'howmany': 'how many', 'whydo': 'why do', 'doI': 'do I', 'theBest': 'the best', 'howdoes': 'how does', 'mastrubation': 'masturbation', 'mastrubate': 'masturbate', \"mastrubating\": 'masturbating', 'pennis': 'penis', 'Etherium': 'Ethereum', 'narcissit': 'narcissist', 'bigdata': 'big data', '2k17': '2017', '2k18': '2018', 'qouta': 'quota', 'exboyfriend': 'ex boyfriend', 'airhostess': 'air hostess', \"whst\": 'what', 'watsapp': 'whatsapp', 'demonitisation': 'demonetization', 'demonitization': 'demonetization', 'demonetisation': 'demonetization'}","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"43b2b0c36dd9dc83000b017406b2ac862d1db84f","scrolled":true},"cell_type":"code","source":"# clear mispells\ndef correct_spelling(x, dic):\n    for word in dic.keys():\n        x = x.replace(word, dic[word])\n    return x\n\ndf['treated_question'] = df['treated_question'].progress_apply(lambda x: correct_spelling(x, mispell_dict))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"891c434be0d55adf01e0cbc2a3c0b1126006abb3"},"cell_type":"code","source":"# Build vocabulary of text data after cleaning mispells\n    ## check coverage of each embedding methods\n# vocab = build_vocab(df['treated_question'])\n# print(\"Glove : \")\n# oov_glove = check_coverage(vocab, embed_glove)\nprint(\"Paragram : \")\n# oov_paragram = check_coverage(vocab, embed_paragram)\n# print(\"FastText : \")\n# oov_fasttext = check_coverage(vocab, embed_fasttext)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d94995a596a50feea85a73cf60a3157ef1fc5041"},"cell_type":"code","source":"# oov_paragram[:100]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"7ef59ab3bc56bb4f2e5203e6546855980ab8442a"},"cell_type":"code","source":"# 'kg' in embed_glove","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a399f3cd218cceb2196fb1e68c2c7e58858107ec"},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"60b3564664acb444d7a2330a00b6ff150082f41c"},"cell_type":"markdown","source":"# (5)Deal with numbers(2 cases)\n- split 165cm to 165 cm\n- replace 165 by ###"},{"metadata":{"trusted":true,"_uuid":"2687860a3c9aa49743856177570fff8d44a58631"},"cell_type":"code","source":"import re\ndef clean_numbers(x):\n    if re.search(\"[0-9]+\", x) != None:\n        x = re.sub('[0-9]+',' {} '.format(re.search('[0-9]+',x).group()),x) \n    x = re.sub('[0-9]{5,}', '#####', x)\n    x = re.sub('[0-9]{4}', '####', x)\n    x = re.sub('[0-9]{3}', '###', x)\n    x = re.sub('[0-9]{2}', '##', x)\n    return x","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d202fd446d092899e63e4749ceb17494e1d2620d"},"cell_type":"code","source":"df[\"treated_question_num\"] = df[\"treated_question\"].progress_apply(lambda x: clean_numbers(x))\n# vocab_num = build_vocab(df[\"treated_question_num\"])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"678570880c41056990fb411b7a3e42ba408444e4"},"cell_type":"code","source":"# Build vocabulary of text data after  numbers\n    ## check coverage of each embedding methods\n# vocab_num = build_vocab(df[\"treated_question_num\"])\n# print(\"Glove : \")\n# oov_glove_num = check_coverage(vocab_num, embed_glove)\nprint(\"Paragram : \")\n# oov_paragram_num = check_coverage(vocab_num, embed_paragram)\n# print(\"FastText : \")\n# oov_fasttext_num = check_coverage(vocab_num, embed_fasttext)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"2dfa3e762f1c30718098260dd57bada36d34f5ab"},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"ed2c814470c052cf3511f1eec2d710a5b854e2c5"},"cell_type":"markdown","source":"# Finished! Using embed_paragram got the best coverage rate!\n- Found embeddings for 74.31% of vocab\n- Found embeddings for  99.66% of all text\n"},{"metadata":{"trusted":true,"scrolled":true,"_uuid":"fc66799dd9fd03302b27d23d26cab8e02e869aa7"},"cell_type":"code","source":"df.columns","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"986baf8e9bd1f005d88f63f25cf63154f793d78a"},"cell_type":"code","source":"# modified from load_glove\ndef build_embedding_matrix(word_index, embed):\n    all_embs = np.stack(embed.values())\n    emb_mean,emb_std = all_embs.mean(), all_embs.std()\n    embed_size = all_embs.shape[1]\n\n    # word_index = tokenizer.word_index\n#     nb_words = min(max_features, len(word_index))\n    nb_words = len(word_index)\n    embedding_matrix = np.random.normal(emb_mean, emb_std, (nb_words, embed_size))\n    for word, i in word_index.items():\n        if i >= nb_words: continue\n        embedding_vector = embed.get(word)\n        if embedding_vector is not None: embedding_matrix[i] = embedding_vector\n    return embedding_matrix \n\n# modified from load_glove\ndef build_embedding_matrix_v2(word_index, embed):\n    all_embs = np.stack(embed.values())\n    emb_mean,emb_std = all_embs.mean(), all_embs.std()\n    embed_size = all_embs.shape[1]\n    \n    empty_vector = np.random.normal(emb_mean, emb_std,(embed_size,))\n    return np.vstack([empty_vector, all_embs])\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d542d690ca0710cd8076a05843c5b8c7702c72ee","scrolled":false},"cell_type":"code","source":"embedding_matrix = build_embedding_matrix_v2(_, embed_paragram)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"3cd16fd193d32cb81c4d201e1c74719ea5697b59"},"cell_type":"code","source":"word_index = {word:i for i,word in enumerate(embed_paragram.keys(),1)}","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"41b0a43a3201f2612e13b139e007c1bd1619019d"},"cell_type":"code","source":"def label_sentence(s, word_index):\n    return [word_index.get(x,0) for x in s.split()]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"710153be5978411ec2cddb411c0a86d1ac7e033b"},"cell_type":"code","source":"train_embed = df[:train.shape[0]][['qid', 'target', 'treated_question_num']]\ntest_embed = df[train.shape[0]:][['qid', 'treated_question_num']]\n\nfrom sklearn.model_selection import train_test_split\ntrain, val = train_test_split(train_embed, test_size=0.1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d34c72201f4a501cef2b96a06e0b2fee5a0f5ca5","scrolled":true},"cell_type":"code","source":"maxlen = 60\nfrom keras.preprocessing.sequence import pad_sequences\ny_train = train['target']\ny_val = val['target']","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"8c3b7ed92a3c4a850f9cec1b71cca54a230833c3"},"cell_type":"code","source":"%%time\ntrain_X = train['treated_question_num'].apply(label_sentence, args=(word_index,))\ntrain_X = pad_sequences(train_X, maxlen=maxlen)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c0d1ea28faa0284c8196d2a273985851ab43bd7d"},"cell_type":"code","source":"%%time\nval_X = val['treated_question_num'].apply(label_sentence, args=(word_index,))\nval_X = pad_sequences(val_X, maxlen=maxlen)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f47f3b5a1172af9356866d9ccaf7f931220a1015"},"cell_type":"code","source":"%%time\nsub_data = test_embed['treated_question_num'].apply(label_sentence, args=(word_index,))\nsub_data = pad_sequences(sub_data, maxlen=maxlen)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"6f4e2ebd020e48f74410b50376f6da903108bcb0"},"cell_type":"code","source":"import tensorflow as tf\nfrom tensorflow import keras\nimport gc\n\ndel df, train_embed, test_embed\ngc.collect()\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ca47e28cdcdd41ea4a05d7cfbe7685e22b3df477","scrolled":false},"cell_type":"code","source":"model = keras.Sequential()\nmodel.add(keras.layers.Embedding(embedding_matrix.shape[0], 300, weights=[embedding_matrix], input_length=maxlen, trainable=False))\nmodel.add(keras.layers.GlobalAveragePooling1D())\nmodel.add(keras.layers.Dense(16, activation=tf.nn.relu))\nmodel.add(keras.layers.Dense(1, activation=tf.nn.sigmoid))\nmodel.summary()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"70a3fec63b58041b01963ec71dba3ff743140f47"},"cell_type":"code","source":"model.compile(optimizer=tf.train.AdamOptimizer(),\n              loss='binary_crossentropy',\n              metrics=['accuracy'])\nearly_stop = keras.callbacks.EarlyStopping(monitor='val_loss',\n                              min_delta=0,\n                              patience=1,\n                              verbose=0, mode='auto')\n\nhistory = model.fit(train_X,\n                y_train,\n                epochs=100,\n                batch_size=1024,\n                validation_data=(val_X, y_val),\n                verbose=1,\n                callbacks=[early_stop,])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"cfbde4bb5bdbc256316f172ce79fc6b74cb8676d","scrolled":true},"cell_type":"code","source":"from sklearn import metrics\ny_pred = model.predict(val_X, batch_size=1024, verbose=1)\nbest_thresh = 0\nbest_score = 0\nfor thresh in np.arange(0.1, 0.9, 0.01):\n    thresh = np.round(thresh, 2)\n    score = metrics.f1_score(y_val, (y_pred>thresh).astype(int))\n    if score>best_score:\n        best_score=score\n        best_thresh=thresh\n    print(\"F1 score at threshold {0} is {1}\".format(thresh, score))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"7fe5dc237f2b0e45fe71e6da066186417d5ee923"},"cell_type":"code","source":"# from keras.models import Sequential\n# from keras.layers import CuDNNLSTM, Dense, Bidirectional","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"dbc9e9acbf82d16374a7a4ffbd846fae21a8cf5a"},"cell_type":"markdown","source":"#  LSTM model"},{"metadata":{"trusted":true,"_uuid":"e35e8ebde7db0b7860af764f5864c0360c72a08e"},"cell_type":"code","source":"# model = Sequential()\n# model.add(Bidirectional(CuDNNLSTM(64, return_sequences=True),\n#                         input_shape=(30, 300)))\n# model.add(Bidirectional(CuDNNLSTM(64)))\n# model.add(Dense(1, activation=\"sigmoid\"))\n\n# model.compile(loss='binary_crossentropy',\n#               optimizer='adam',\n#               metrics=['accuracy'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"6f7b49e1c90de40f70ec2e9eac72308b55e4c7e2"},"cell_type":"code","source":"# mg = batch_gen(train_df)\n# model.fit_generator(mg, epochs=20,\n#                     steps_per_epoch=1000,\n#                     validation_data=(val_vects, val_y),\n#                     verbose=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"1a33ef90affcde12f8992bdcabf36e2a18da328c"},"cell_type":"code","source":"pred_val_y = model.predict([sub_data], batch_size=1024, verbose=0)\nsub = pd.read_csv('../input/sample_submission.csv')\nsub.prediction = pred_val_y > best_thresh\nsub.to_csv(\"submission.csv\", index=False)","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}