{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nfrom tqdm import tqdm_notebook as tqdm\nimport operator\nfrom nltk.corpus import stopwords\nstopwords  = stopwords.words('english')\nimport re\nfrom nltk.tokenize import word_tokenize\n\nfrom sklearn.model_selection import train_test_split\nfrom sklearn import metrics \n\nfrom keras.preprocessing.text import Tokenizer\nfrom keras.preprocessing.sequence import pad_sequences\nfrom keras.layers import Dense, Input, LSTM, Embedding, Dropout, Activation, CuDNNGRU, Conv1D\nfrom keras.layers import Bidirectional, GlobalMaxPool1D\nfrom keras.models import Model\nfrom keras import initializers, regularizers, constraints, optimizers, layers\nfrom keras.callbacks import EarlyStopping","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"train = pd.read_csv(\"../input/train.csv\")\ntest = pd.read_csv(\"../input/test.csv\")\n\nprint('Train shape : ',train.shape[0])\nprint('Test shape : ',test.shape[0])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"79cfbde54dc3a3af08d28fb6d801695d83834e38"},"cell_type":"code","source":"EMBEDDING_FILE = '../input/embeddings/glove.840B.300d/glove.840B.300d.txt'\ndef get_coefs(word,*arr): return word, np.asarray(arr, dtype='float32')\nembed_index = dict(get_coefs(*o.split(\" \")) for o in open(EMBEDDING_FILE))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"eaa6b9b8e7f0f46bcde7af796e5c423b72bfd7b9"},"cell_type":"code","source":"def build_vocab(texts):\n    vocab = {}\n    sentences = texts.apply(lambda x : x.split()).values\n    for sentence in tqdm(sentences, disable = False):\n        for word in sentence:\n            vocab[word] = vocab.get(word, 0) + 1\n    return vocab   ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"2c7e7d7c710e58db797d8a0076da8d862f276ad6"},"cell_type":"code","source":"vocab  = build_vocab(train['question_text'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"66e4f775065a2978ef3244c86a766609dfb7baf7"},"cell_type":"code","source":"print({k: vocab[k] for k in list(vocab)[:5]})","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"1af3143506c50fd4f1e8b7a32950dc14dc7759ae"},"cell_type":"code","source":"ei = set(embed_index.keys())\ndef check_coverage(vocab, embed_index):\n    known_words = {}\n    unknown_words = {}\n    num_known_words = 0\n    num_unknown_words = 0\n    for word in tqdm(list(vocab.keys()), disable = False):\n        if word in ei:\n            known_words[word] = embed_index[word]\n            num_known_words += vocab[word]\n        elif word.lower() in ei:\n            known_words[word.lower()] = embed_index[word.lower()]\n            num_known_words += vocab[word]\n        else:\n            unknown_words[word] = vocab[word]\n            num_unknown_words += vocab[word]\n    print('{:.2%} of words of vocab are known.'.format(len(known_words)/len(vocab)))\n    print('{:.2%} of all text is known.'.format(num_known_words/(num_known_words + num_unknown_words),2))\n    unknown_words = sorted(unknown_words.items(), key=operator.itemgetter(1))[::-1]\n    return known_words, unknown_words","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"5a76839810c3c17b63fec5ac7c674d900987e19a"},"cell_type":"code","source":"kw, uw = check_coverage(vocab, embed_index)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"77430b31593631beca8bebe6359135c6091b17b0"},"cell_type":"code","source":"uw[:10]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a7ff3f8df97a1f4a7ce6f1a9d85f542afbcf7af0"},"cell_type":"code","source":"'?' in ei","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"36348ef279167346e4340d4c6f14e526dcae4b79"},"cell_type":"code","source":"train['question_text'] = train['question_text'].apply(lambda x : x.replace('?',''))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"77bc5e3c6bcef7ff16e4323c8554e4c4af418338"},"cell_type":"code","source":"vocab  = build_vocab(train['question_text'])\nprint({k: vocab[k] for k in list(vocab)[:5]})","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"537806d305e801897e8c7fa0c5f040f0f1bf68c6"},"cell_type":"code","source":"kw, uw = check_coverage(vocab, embed_index)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"3dd100e1f6a3ba8a494138956c9abc4495fd6c5b"},"cell_type":"code","source":"uw[:20]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e70de1cd8040f7414ee740644fa8a0e2535cfff2"},"cell_type":"code","source":"def clean_numbers(sentence):\n    sentence = re.sub('[0-9]{5,}','#####', sentence)\n    sentence = re.sub('[0-9]{4}','####', sentence)\n    sentence = re.sub('[0-9]{3}','###', sentence)\n    sentence = re.sub('[0-9]{2}','##', sentence)\n    sentence = re.sub('[0-9]{1}','#', sentence)\n    return sentence\n    ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"cad8a44d2a3487c5b40fd21912c94f2abcbd543f"},"cell_type":"code","source":"train['question_text'] = train['question_text'].apply(lambda x : clean_numbers(x))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d4b0860b543447d23b39cb84f37832231fc0d404"},"cell_type":"code","source":"vocab  = build_vocab(train['question_text'])\nkw, uw = check_coverage(vocab, embed_index)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"79d665c0246f2e015fe343200c0fcaf06ae19786"},"cell_type":"code","source":"print(uw[:20])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"6f3cfd3c7862847e4ff62810e4c309da00c5b1bf"},"cell_type":"code","source":"train['question_text'] = train['question_text'].apply(lambda x : x.replace(\"/\",\" \"))\ntrain['question_text'] = train['question_text'].apply(lambda x : x.replace(\"-\",\" \"))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"928575665985660c3565c0b0b2943eab776a6bf5"},"cell_type":"code","source":"vocab  = build_vocab(train['question_text'])\nkw, uw = check_coverage(vocab, embed_index)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a3419b164be09a0ec9c234d8fef67934a48c540b"},"cell_type":"code","source":"uw[:30]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"91a4fd6eef4e949b5c0e44569f3d3cbcd1d013c8"},"cell_type":"code","source":"def clean_punctuations(sentence):\n    sentence = str(sentence)\n    for punct in '&':\n        sentence = sentence.replace('&', f' {punct} ')\n    for punct in '?!.,#$%\\()*+-/:;<=>@[\\\\]^_{|}~\"':\n        sentence = sentence.replace(punct, '')\n    return sentence","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"df653b69d39a960c5e5cb7b9cff5b93adb566c59"},"cell_type":"code","source":"train['question_text'] = train['question_text'].apply(lambda x : clean_punctuations(x))\nvocab  = build_vocab(train['question_text'])\nkw, uw = check_coverage(vocab, embed_index)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"41db19fe2e5757dd26f1b892b98c705a61407a0c"},"cell_type":"code","source":"uw[:30]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c7752dc746ebea7029ea6d7bf7229ce52e59b23d"},"cell_type":"code","source":"specials = [\"’\", \"‘\", \"´\", \"`\"]\ndef clean_apostrophe(sentence):\n    for s in specials:\n        sentence = sentence.replace(s,\"'\")\n    return sentence    ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d9cd57d29f3484b5e2cfe2b03f051f99364d38ef"},"cell_type":"code","source":"train['question_text'] = train['question_text'].apply(lambda x : clean_apostrophe(x))\nvocab  = build_vocab(train['question_text'])\nkw, uw = check_coverage(vocab, embed_index)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c1d6667a0a5242875e3840cb8cbf6887bf498d69"},"cell_type":"code","source":"uw[:20]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"316c46f614baa9d61a4465f724ea49b09d554c85"},"cell_type":"code","source":"contraction_mapping = {\"ain't\": \"is not\", \"aren't\": \"are not\",\"can't\": \"cannot\", \"'cause\": \"because\", \"could've\": \"could have\", \"couldn't\": \"could not\", \"didn't\": \"did not\",  \"doesn't\": \"does not\", \"don't\": \"do not\", \"hadn't\": \"had not\", \"hasn't\": \"has not\", \"haven't\": \"have not\", \"he'd\": \"he would\",\"he'll\": \"he will\", \"he's\": \"he is\", \"how'd\": \"how did\", \"how'd'y\": \"how do you\", \"how'll\": \"how will\", \"how's\": \"how is\",  \"I'd\": \"I would\", \"I'd've\": \"I would have\", \"I'll\": \"I will\", \"I'll've\": \"I will have\",\"I'm\": \"I am\", \"I've\": \"I have\", \"i'd\": \"i would\", \"i'd've\": \"i would have\", \"i'll\": \"i will\",  \"i'll've\": \"i will have\",\"i'm\": \"i am\", \"i've\": \"i have\", \"isn't\": \"is not\", \"it'd\": \"it would\", \"it'd've\": \"it would have\", \"it'll\": \"it will\", \"it'll've\": \"it will have\",\"it's\": \"it is\", \"let's\": \"let us\", \"ma'am\": \"madam\", \"mayn't\": \"may not\", \"might've\": \"might have\",\"mightn't\": \"might not\",\"mightn't've\": \"might not have\", \"must've\": \"must have\", \"mustn't\": \"must not\", \"mustn't've\": \"must not have\", \"needn't\": \"need not\", \"needn't've\": \"need not have\",\"o'clock\": \"of the clock\", \"oughtn't\": \"ought not\", \"oughtn't've\": \"ought not have\", \"shan't\": \"shall not\", \"sha'n't\": \"shall not\", \"shan't've\": \"shall not have\", \"she'd\": \"she would\", \"she'd've\": \"she would have\", \"she'll\": \"she will\", \"she'll've\": \"she will have\", \"she's\": \"she is\", \"should've\": \"should have\", \"shouldn't\": \"should not\", \"shouldn't've\": \"should not have\", \"so've\": \"so have\",\"so's\": \"so as\", \"this's\": \"this is\",\"that'd\": \"that would\", \"that'd've\": \"that would have\", \"that's\": \"that is\", \"there'd\": \"there would\", \"there'd've\": \"there would have\", \"there's\": \"there is\", \"here's\": \"here is\",\"they'd\": \"they would\", \"they'd've\": \"they would have\", \"they'll\": \"they will\", \"they'll've\": \"they will have\", \"they're\": \"they are\", \"they've\": \"they have\", \"to've\": \"to have\", \"wasn't\": \"was not\", \"we'd\": \"we would\", \"we'd've\": \"we would have\", \"we'll\": \"we will\", \"we'll've\": \"we will have\", \"we're\": \"we are\", \"we've\": \"we have\", \"weren't\": \"were not\", \"what'll\": \"what will\", \"what'll've\": \"what will have\", \"what're\": \"what are\",  \"what's\": \"what is\", \"what've\": \"what have\", \"when's\": \"when is\", \"when've\": \"when have\", \"where'd\": \"where did\", \"where's\": \"where is\", \"where've\": \"where have\", \"who'll\": \"who will\", \"who'll've\": \"who will have\", \"who's\": \"who is\", \"who've\": \"who have\", \"why's\": \"why is\", \"why've\": \"why have\", \"will've\": \"will have\", \"won't\": \"will not\", \"won't've\": \"will not have\", \"would've\": \"would have\", \"wouldn't\": \"would not\", \"wouldn't've\": \"would not have\", \"y'all\": \"you all\", \"y'all'd\": \"you all would\",\"y'all'd've\": \"you all would have\",\"y'all're\": \"you all are\",\"y'all've\": \"you all have\",\"you'd\": \"you would\", \"you'd've\": \"you would have\", \"you'll\": \"you will\", \"you'll've\": \"you will have\", \"you're\": \"you are\", \"you've\": \"you have\", 'colour': 'color', 'centre': 'center', 'favourite': 'favorite', 'travelling': 'traveling', 'counselling': 'counseling', 'theatre': 'theater', 'cancelled': 'canceled', 'labour': 'labor', 'organisation': 'organization', 'wwii': 'world war 2', 'citicise': 'criticize', 'youtu ': 'youtube ', 'Qoura': 'Quora', 'sallary': 'salary', 'Whta': 'What', 'narcisist': 'narcissist', 'howdo': 'how do', 'whatare': 'what are', 'howcan': 'how can', 'howmuch': 'how much', 'howmany': 'how many', 'whydo': 'why do', 'doI': 'do I', 'theBest': 'the best', 'howdoes': 'how does', 'mastrubation': 'masturbation', 'mastrubate': 'masturbate', \"mastrubating\": 'masturbating', 'pennis': 'penis', 'Etherium': 'Ethereum', 'narcissit': 'narcissist', 'bigdata': 'big data', '2k17': '2017', '2k18': '2018', 'qouta': 'quota', 'exboyfriend': 'ex boyfriend', 'airhostess': 'air hostess', \"whst\": 'what', 'watsapp': 'whatsapp', 'demonitisation': 'demonetization', 'demonitization': 'demonetization', 'demonetisation': 'demonetization'}","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"6352ef0b57cf6524ddc5216a9fb5a29291579ec4"},"cell_type":"code","source":"def remove_stopwords(sentence):\n    words = sentence.split(\" \")\n    words = [contraction_mapping[word.lower()] if word.lower() in contraction_mapping.keys() else word for word in words]\n    return \" \".join(words)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"4f66e3023b68bf223f094fa7cd179ef740316292"},"cell_type":"code","source":"train['question_text'] = train['question_text'].apply(lambda x : remove_stopwords(x))\nvocab  = build_vocab(train['question_text'])\nkw, uw = check_coverage(vocab, embed_index)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"dc9b2502f928562818410472b1e44ee5fc0ee6b9"},"cell_type":"code","source":"uw[:30]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"5d50251b9141c5298b88a708e97d4305186fb44c"},"cell_type":"code","source":"train['question_text'] = train['question_text'].apply(lambda x: re.sub(\"'s\", ' is', x))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"3da975fe2fb7fa8547ccc31c6ad6a7a4cb5f520b"},"cell_type":"code","source":"vocab  = build_vocab(train['question_text'])\nkw, uw = check_coverage(vocab, embed_index)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"be6f93a5e0f448041ade51830200765e93289d6b"},"cell_type":"code","source":"uw[:20]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a37d1e19ab93ffe36f3e87d32453b4c7001bb412"},"cell_type":"code","source":"train['question_text'] = train['question_text'].apply(lambda x: x.replace('Quorans', 'Quora'))\ntrain['question_text'] = train['question_text'].apply(lambda x: x.replace('Brexit', 'Europe'))\ntrain['question_text'] = train['question_text'].apply(lambda x: x.replace('₹', 'rupee'))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a7ff25eda81043adb4cca60c33247fc8bea19d7a"},"cell_type":"code","source":"'bible' in embed_index","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a802151ff0f873a0c103c6ec618bfb39a3053a1b"},"cell_type":"code","source":"vocab  = build_vocab(train['question_text'])\nkw, uw = check_coverage(vocab, embed_index)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"54f8777f6ac60dae6b8b7743fb2cefb63f5e6fd9","scrolled":false},"cell_type":"code","source":"uw[:20]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"37dd42d8a394d6d0bcf9184b52f589da5b68e6e9","scrolled":true},"cell_type":"code","source":"'altcoin.com' in embed_index","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b7cd8b61fff24acc2a611cd98640ed26e7782fb9"},"cell_type":"code","source":"train['question_text'] = train['question_text'].apply(lambda x: x.replace(\"'\", ''))\ntrain['question_text'] = train['question_text'].apply(lambda x: x.replace('“', ''))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a9a1068ebf06eb81602326be9670c029d96b2429"},"cell_type":"code","source":"vocab  = build_vocab(train['question_text'])\nkw, uw = check_coverage(vocab, embed_index)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"856c001a2d5c6ab0757bf27860050b3752ef7659"},"cell_type":"code","source":"uw[:20]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"6f64e4a9ff2622a8acf0bae4c8e8e11da705b3fb"},"cell_type":"code","source":"train['question_text'] = train['question_text'].apply(lambda x: x.replace('cryptocurrencies', 'cryptocurrency'))\ntrain['question_text'] = train['question_text'].apply(lambda x: x.replace('Redmi', 'mobile company'))\ntrain['question_text'] = train['question_text'].apply(lambda x: x.replace('OnePlus', 'mobile company'))\ntrain['question_text'] = train['question_text'].apply(lambda x: x.replace('Qoura', 'Quora'))\ntrain['question_text'] = train['question_text'].apply(lambda x: x.replace('°C', 'degree celsius'))\ntrain['question_text'] = train['question_text'].apply(lambda x: x.replace('etc…', 'etc.'))\ntrain['question_text'] = train['question_text'].apply(lambda x: x.replace('Bhakts', 'supporter'))\ntrain['question_text'] = train['question_text'].apply(lambda x: x.replace('bhakts', 'supporter'))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"8ccf99b54ae709ac25718402498d9853e2b51a8e"},"cell_type":"code","source":"vocab  = build_vocab(train['question_text'])\nkw, uw = check_coverage(vocab, embed_index)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"75a44c13ea00e1958c5bb061ef5fea5d049d1201"},"cell_type":"code","source":"uw[:20]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"1152775bc943cec073f6ca562ed7d92d451709dc"},"cell_type":"code","source":"test['question_text'] = test['question_text'].apply(lambda x : x.replace(\"/\",\" \"))\ntest['question_text'] = test['question_text'].apply(lambda x : x.replace(\"-\",\" \"))\n\ntest['question_text'] = test['question_text'].apply(lambda x : clean_numbers(x))\ntest['question_text'] = test['question_text'].apply(lambda x : clean_punctuations(x))\n\ntest['question_text'] = test['question_text'].apply(lambda x : clean_apostrophe(x))\n\ntest['question_text'] = test['question_text'].apply(lambda x : remove_stopwords(x))\n\ntest['question_text'] = test['question_text'].apply(lambda x : re.sub(\"'s\", ' is', x))\n\ntest['question_text'] = test['question_text'].apply(lambda x: x.replace(\"'\", ''))\ntest['question_text'] = test['question_text'].apply(lambda x: x.replace('“', ''))\n\ntest['question_text'] = test['question_text'].apply(lambda x: x.replace('Quorans', 'Quora'))\ntest['question_text'] = test['question_text'].apply(lambda x: x.replace('Brexit', 'Europe'))\ntest['question_text'] = test['question_text'].apply(lambda x: x.replace('₹', 'rupee'))\ntest['question_text'] = test['question_text'].apply(lambda x: x.replace('cryptocurrencies', 'cryptocurrency'))\ntest['question_text'] = test['question_text'].apply(lambda x: x.replace('Redmi', 'mobile company'))\ntest['question_text'] = test['question_text'].apply(lambda x: x.replace('OnePlus', 'mobile company'))\ntest['question_text'] = test['question_text'].apply(lambda x: x.replace('Qoura', 'Quora'))\ntest['question_text'] = test['question_text'].apply(lambda x: x.replace('°C', 'degree celsius'))\ntest['question_text'] = test['question_text'].apply(lambda x: x.replace('etc…', 'etc.'))\ntest['question_text'] = test['question_text'].apply(lambda x: x.replace('Bhakts', 'supporter'))\ntest['question_text'] = test['question_text'].apply(lambda x: x.replace('bhakts', 'supporter'))\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ece340591ff3d61377294bd6847b66756b75a7c1"},"cell_type":"code","source":"X_train = train[\"question_text\"].values\ny = pd.read_csv('../input/train.csv')['target']\nX_test = test[\"question_text\"].values\n\nx_train, x_val, y_train, y_val = train_test_split(X_train, y , test_size = 0.2)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"58355b163cdf92abea64b09f87b038bb5b95c43b"},"cell_type":"code","source":"all_embed = np.stack(embed_index.values())\nprint(all_embed.shape)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"8239732663b3d780d6eaa56e82c7849074dc7481"},"cell_type":"code","source":"emb_mean, emb_std = all_embed.mean(), all_embed.std()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"312682eda8f01870568560f0062d4944acf3bf4c"},"cell_type":"code","source":"tokenizer = Tokenizer(num_words = 50000)\ntokenizer.fit_on_texts(list(x_train))\nx_train = tokenizer.texts_to_sequences(x_train)\nx_val = tokenizer.texts_to_sequences(x_val)\nx_test = tokenizer.texts_to_sequences(X_test)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"882dae8d6ff5f0221a2272cd729f27c23ae58935"},"cell_type":"code","source":"del train, test","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"7ac53e380dd016e940ffdab411fc2a8c116e995e"},"cell_type":"code","source":"del X_train, X_test","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ea1c72fed7f0e46ba37068d3e28cc3da941d0d90"},"cell_type":"code","source":"maxlen = 100\nx_train = pad_sequences(x_train, maxlen = maxlen)\nx_val = pad_sequences(x_val, maxlen = maxlen)\nx_test = pad_sequences(x_test, maxlen = maxlen)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d5af8e60b29af7bb60b1779ead18a8832d9ad690"},"cell_type":"code","source":"word_index = tokenizer.word_index\nnb_words = min(50000, len(word_index))\nprint(nb_words)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"7a22494875d067c717e0de17c0076202ce8081e6"},"cell_type":"code","source":"embed_matrix = np.random.normal(emb_mean, emb_std, (nb_words, 300))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f01e420b85dbe42d97e253e16986cdf2f0d68f5b"},"cell_type":"code","source":"for word, i in word_index.items():\n    if i >= 50000: continue\n    embed_vector = embed_index.get(word)\n    if embed_vector is not None:\n        embed_matrix[i] = embed_vector","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c5ab9883ac8ec7be9305c7495c79fd1f6e845f24"},"cell_type":"code","source":"def get_model():\n    inp = Input(shape = (maxlen,))\n    x = Embedding(50000, 300, weights = [embed_matrix])(inp)\n    x = Bidirectional(LSTM(64, return_sequences=True))(x)\n    x = GlobalMaxPool1D()(x)\n    x = Dense(16, activation='relu')(x)\n    x = Dropout(0.1)(x)\n    x = Dense(1, activation='sigmoid')(x)\n    model = Model(inputs = inp, outputs = x)\n    model.compile(loss = 'binary_crossentropy', optimizer = 'adam', metrics = ['accuracy'])\n    return model","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"79ad089379cab13dcfa33dfe66b0b2919ad71b4e"},"cell_type":"code","source":"model = get_model()\n\nprint(model.summary())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"9b38cb49a8cc7059c3ff5bb3fd2e6b3e97df4271"},"cell_type":"code","source":"model.fit(x_train, y_train, batch_size = 512, epochs = 3, validation_data = (x_val, y_val), callbacks = [EarlyStopping(monitor='val_loss', min_delta = 0.0001)])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"58d85c4b9118fa508ffc0775995ddb2b22c32f57"},"cell_type":"code","source":"y_valpred = model.predict(x_val)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"9df09071a05ad0ed168ab4bc6a376dc7279ee882"},"cell_type":"code","source":"f1 = 0\nthreshold = 0\n\nfor thresh in np.arange(0.1, 0.501,0.01):\n    f1score = np.round(metrics.f1_score(y_val, (y_valpred>thresh).astype(int)), 4)\n    thresh = np.round(thresh,2)\n    print('F1 score for threshold {} : {}'.format(thresh, f1score))\n    if f1score > f1:\n        f1 = f1score\n        threshold = thresh\n        #print('In {} : {}'.format(threshold,f1score))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"20a908863d0f40931810cb182ee28c1939ddaf5b"},"cell_type":"code","source":"print(threshold)\nprint(f1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a00c35da33dd2292e1516d43540d39b7c6c2e36e"},"cell_type":"code","source":"y_test = model.predict(x_test)\ny_test = (y_test[:,0] > threshold).astype(np.int)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"9332983bb4edbfae96ceebf79e99f6d44b2f32e7"},"cell_type":"code","source":"test = pd.read_csv(\"../input/test.csv\")['qid']","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"868591ea58ab1526647f4e4932002a177557242d"},"cell_type":"code","source":"submit_df = pd.DataFrame({\"qid\": test, \"prediction\": y_test})\nsubmit_df.to_csv(\"submission.csv\", index=False)","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}