{"cells":[{"metadata":{"_uuid":"08215cc1b261c9152b5d7763502d632a0e4446f6"},"cell_type":"markdown","source":"### Cleaning Text\nHere we do things like remove punctuation (specifically ? marks), misspellings, and more"},{"metadata":{"trusted":true,"_uuid":"fef832e2524bd342f29bb522f969741bec5978a1"},"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport operator \nimport re\nimport gc\nimport keras\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nsns.set_style('whitegrid')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"45de53402a6cabe44039d47f7cf5dfcef384b203"},"cell_type":"code","source":"### IMPORTANT HELPER FUNCTIONS\ndef preprocess_text(df):\n    \"\"\"\n    Processes a question and puts the results in the new column 'treated_question'\n    \"\"\"\n    df['treated_question'] = df['question_text'].apply(lambda x: x.lower())\n    df['treated_question'] = df['treated_question'].apply(lambda x: clean_contractions(x, contraction_mapping))\n    df['treated_question'] = df['treated_question'].apply(lambda x: clean_special_chars(x, punct, punct_mapping))\n    df['treated_question'] = df['treated_question'].apply(lambda x: correct_spelling(x, mispell_dict))\n\ndef load_embed(file):\n    \"\"\"\n    Loads a word embedding from a file into a dictionary in the format of {word: word_embedding_vector}\n    \"\"\"\n    def get_coefs(word,*arr): \n        return word, np.asarray(arr, dtype='float32')\n    \n    if file == '../input/embeddings/wiki-news-300d-1M/wiki-news-300d-1M.vec':\n        embeddings_index = dict(get_coefs(*o.split(\" \")) for o in open(file) if len(o)>100)\n    else:\n        embeddings_index = dict(get_coefs(*o.split(\" \")) for o in open(file, encoding='latin'))\n        \n    return embeddings_index\n\ndef build_vocab(texts):\n    \"\"\"\n    Creates a vocabulary from a source text\n    \"\"\"\n    sentences = texts.apply(lambda x: x.split()).values\n    vocab = {}\n    for sentence in sentences:\n        for word in sentence:\n            try:\n                vocab[word] += 1\n            except KeyError:\n                vocab[word] = 1\n    return vocab\n\ndef check_coverage(vocab, embeddings_index):\n    \"\"\"\n    Checks what percentages of vocabulary and source text\n    are covered by our embedding\n    \"\"\"\n    known_words = {}\n    unknown_words = {}\n    nb_known_words = 0\n    nb_unknown_words = 0\n    for word in vocab.keys():\n        try:\n            known_words[word] = embeddings_index[word]\n            nb_known_words += vocab[word]\n        except:\n            unknown_words[word] = vocab[word]\n            nb_unknown_words += vocab[word]\n            pass\n\n    print('Found embeddings for {:.3%} of vocab'.format(len(known_words) / len(vocab)))\n    print('Found embeddings for  {:.3%} of all text'.format(nb_known_words / (nb_known_words + nb_unknown_words)))\n    unknown_words = sorted(unknown_words.items(), key=operator.itemgetter(1))[::-1]\n\n    return unknown_words\n\ndef add_lower(embedding, vocab):\n    \"\"\" \n    Adds the corresponding lowercase version of every word that is \n    only in our embedding as an uppercase word\n    \"\"\"\n    count = 0\n    for word in vocab:\n        if word in embedding and word.lower() not in embedding:  \n            embedding[word.lower()] = embedding[word]\n            count += 1\n    print(f\"Added {count} words to embedding\")\n    \n# A mapping of common contractions to their non-contraction counterparts\ncontraction_mapping = {\"ain't\": \"is not\", \"aren't\": \"are not\",\"can't\": \"cannot\", \"'cause\": \"because\", \"could've\": \"could have\", \"couldn't\": \"could not\", \"didn't\": \"did not\",  \"doesn't\": \"does not\", \"don't\": \"do not\", \"hadn't\": \"had not\", \"hasn't\": \"has not\", \"haven't\": \"have not\", \"he'd\": \"he would\",\"he'll\": \"he will\", \"he's\": \"he is\", \"how'd\": \"how did\", \"how'd'y\": \"how do you\", \"how'll\": \"how will\", \"how's\": \"how is\",  \"I'd\": \"I would\", \"I'd've\": \"I would have\", \"I'll\": \"I will\", \"I'll've\": \"I will have\",\"I'm\": \"I am\", \"I've\": \"I have\", \"i'd\": \"i would\", \"i'd've\": \"i would have\", \"i'll\": \"i will\",  \"i'll've\": \"i will have\",\"i'm\": \"i am\", \"i've\": \"i have\", \"isn't\": \"is not\", \"it'd\": \"it would\", \"it'd've\": \"it would have\", \"it'll\": \"it will\", \"it'll've\": \"it will have\",\"it's\": \"it is\", \"let's\": \"let us\", \"ma'am\": \"madam\", \"mayn't\": \"may not\", \"might've\": \"might have\",\"mightn't\": \"might not\",\"mightn't've\": \"might not have\", \"must've\": \"must have\", \"mustn't\": \"must not\", \"mustn't've\": \"must not have\", \"needn't\": \"need not\", \"needn't've\": \"need not have\",\"o'clock\": \"of the clock\", \"oughtn't\": \"ought not\", \"oughtn't've\": \"ought not have\", \"shan't\": \"shall not\", \"sha'n't\": \"shall not\", \"shan't've\": \"shall not have\", \"she'd\": \"she would\", \"she'd've\": \"she would have\", \"she'll\": \"she will\", \"she'll've\": \"she will have\", \"she's\": \"she is\", \"should've\": \"should have\", \"shouldn't\": \"should not\", \"shouldn't've\": \"should not have\", \"so've\": \"so have\",\"so's\": \"so as\", \"this's\": \"this is\",\"that'd\": \"that would\", \"that'd've\": \"that would have\", \"that's\": \"that is\", \"there'd\": \"there would\", \"there'd've\": \"there would have\", \"there's\": \"there is\", \"here's\": \"here is\",\"they'd\": \"they would\", \"they'd've\": \"they would have\", \"they'll\": \"they will\", \"they'll've\": \"they will have\", \"they're\": \"they are\", \"they've\": \"they have\", \"to've\": \"to have\", \"wasn't\": \"was not\", \"we'd\": \"we would\", \"we'd've\": \"we would have\", \"we'll\": \"we will\", \"we'll've\": \"we will have\", \"we're\": \"we are\", \"we've\": \"we have\", \"weren't\": \"were not\", \"what'll\": \"what will\", \"what'll've\": \"what will have\", \"what're\": \"what are\",  \"what's\": \"what is\", \"what've\": \"what have\", \"when's\": \"when is\", \"when've\": \"when have\", \"where'd\": \"where did\", \"where's\": \"where is\", \"where've\": \"where have\", \"who'll\": \"who will\", \"who'll've\": \"who will have\", \"who's\": \"who is\", \"who've\": \"who have\", \"why's\": \"why is\", \"why've\": \"why have\", \"will've\": \"will have\", \"won't\": \"will not\", \"won't've\": \"will not have\", \"would've\": \"would have\", \"wouldn't\": \"would not\", \"wouldn't've\": \"would not have\", \"y'all\": \"you all\", \"y'all'd\": \"you all would\",\"y'all'd've\": \"you all would have\",\"y'all're\": \"you all are\",\"y'all've\": \"you all have\",\"you'd\": \"you would\", \"you'd've\": \"you would have\", \"you'll\": \"you will\", \"you'll've\": \"you will have\", \"you're\": \"you are\", \"you've\": \"you have\" }\n\ndef known_contractions(embed):\n    \"\"\"\n    Prints what contractions an embedding was trained on\n    \"\"\"\n    known = []\n    for contract in contraction_mapping:\n        if contract in embed:\n            known.append(contract)\n    return known\n\ndef clean_contractions(text, mapping):\n    \"\"\"\n    Replces all strange apostrophy characters with the standard ascii apostrophe\n    \"\"\"\n    specials = [\"’\", \"‘\", \"´\", \"`\"]\n    for s in specials:\n        text = text.replace(s, \"'\")\n    text = ' '.join([mapping[t] if t in mapping else t for t in text.split(\" \")])\n    return text\n\n# A string of common punctuation and strange characters\npunct = \"/-'?!.,#$%\\'()*+-/:;<=>@[\\\\]^_`{|}~\" + '\"\"“”’' + '∞θ÷α•à−β∅³π‘₹´°£€\\×™√²—–&'\n# A mapping of strange characters to their common ascii counterparts\npunct_mapping = {\"‘\": \"'\", \"₹\": \"e\", \"´\": \"'\", \"°\": \"\", \"€\": \"e\", \"™\": \"tm\", \"√\": \" sqrt \", \"×\": \"x\", \"²\": \"2\", \"—\": \"-\", \"–\": \"-\", \"’\": \"'\", \"_\": \"-\", \"`\": \"'\", '“': '\"', '”': '\"', '“': '\"', \"£\": \"e\", '∞': 'infinity', 'θ': 'theta', '÷': '/', 'α': 'alpha', '•': '.', 'à': 'a', '−': '-', 'β': 'beta', '∅': '', '³': '3', 'π': 'pi', }\n\ndef unknown_punct(embed, punct):\n    unknown = ''\n    for p in punct:\n        if p not in embed:\n            unknown += p\n            unknown += ' '\n    return unknown\n\ndef clean_special_chars(text, punct, mapping):\n    for p in mapping:\n        text = text.replace(p, mapping[p])\n    \n    for p in punct:\n        text = text.replace(p, f' {p} ')\n    \n    specials = {'\\u200b': ' ', '…': ' ... ', '\\ufeff': '', 'करना': '', 'है': ''}  # Other special characters that I have to deal with in last\n    for s in specials:\n        text = text.replace(s, specials[s])\n    \n    return text\n\nmispell_dict = {'colour': 'color', 'centre': 'center', 'favourite': 'favorite', 'travelling': 'traveling', 'counselling': 'counseling', 'theatre': 'theater', 'cancelled': 'canceled', 'labour': 'labor', 'organisation': 'organization', 'wwii': 'world war 2', 'citicise': 'criticize', 'youtu ': 'youtube ', 'Qoura': 'Quora', 'sallary': 'salary', 'Whta': 'What', 'narcisist': 'narcissist', 'howdo': 'how do', 'whatare': 'what are', 'howcan': 'how can', 'howmuch': 'how much', 'howmany': 'how many', 'whydo': 'why do', 'doI': 'do I', 'theBest': 'the best', 'howdoes': 'how does', 'mastrubation': 'masturbation', 'mastrubate': 'masturbate', \"mastrubating\": 'masturbating', 'pennis': 'penis', 'Etherium': 'Ethereum', 'narcissit': 'narcissist', 'bigdata': 'big data', '2k17': '2017', '2k18': '2018', 'qouta': 'quota', 'exboyfriend': 'ex boyfriend', 'airhostess': 'air hostess', \"whst\": 'what', 'watsapp': 'whatsapp', 'demonitisation': 'demonetization', 'demonitization': 'demonetization', 'demonetisation': 'demonetization', 'pokémon': 'pokemon'}\n\ndef correct_spelling(x, dic):\n    for word in dic.keys():\n        x = x.replace(word, dic[word])\n    return x\n\ndef print_top_unk_words(unk_list):\n    print('\\nTop unknown words:')\n    print('\\n'.join(map(str, unk_list[:10])))\n","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"f6aa469c99cbfe05444be9e82f65195f938d0a26"},"cell_type":"markdown","source":"### Here we actually *do* the preprocessing"},{"metadata":{"trusted":true,"_uuid":"81a5e03ab72951cf968d3ac97589a6407733c634"},"cell_type":"code","source":"print('Load Train + Test CSVs')\n\ntrain = pd.read_csv(\"../input/train.csv\")\ntest = pd.read_csv(\"../input/test.csv\")\ndf = pd.concat([train ,test])\nprint(\"Number of texts: \", df.shape[0])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"557727411af7f93f438794c4117eb168b76ffa61"},"cell_type":"code","source":"print(\"Extracting GloVe embedding\")\n\nglove = '../input/embeddings/glove.840B.300d/glove.840B.300d.txt'\nembed_glove = load_embed(glove)\nvocab = build_vocab(df['question_text'])","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"fa8f61d9f581dafd21ad76214cb4b3406d98fcf0"},"cell_type":"markdown","source":"## Training a model with our preprocessing pipeline\nNow that we have found a way to improve our model, we can add this in to our training script."},{"metadata":{"trusted":true,"_uuid":"5c245dbb592504dfd6891ff48267f35703a50da3"},"cell_type":"code","source":"from keras.preprocessing.text import Tokenizer\nfrom keras.preprocessing.sequence import pad_sequences\n\nlen_voc = 95000\nmax_len = 60\n\ndef make_data(X, tokenizer):\n    tokenizer.fit_on_texts(X)\n    X = tokenizer.texts_to_sequences(X)\n    X = pad_sequences(X, maxlen=max_len)\n    return X, tokenizer.word_index","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"4b4f717cc944453ab8934584f8eca780baca3f33"},"cell_type":"code","source":"X, word_index = make_data(train['question_text'], Tokenizer(num_words=len_voc))\nX_treated, word_index_treated = make_data(train['question_text'], Tokenizer(num_words=len_voc))\n\npreprocess_text(train)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"55a1d0297a247440f41f858f43967833502404d5"},"cell_type":"markdown","source":"### Model + Train / Test Set Preparation"},{"metadata":{"trusted":true,"_uuid":"287227fdf628c1f3ecca849285a65a473071592d"},"cell_type":"code","source":"from keras import backend as K\nfrom keras.models import Model\nfrom keras.layers import Dense, Embedding, Bidirectional, CuDNNGRU, GlobalAveragePooling1D, GlobalMaxPooling1D, concatenate, Input, Dropout\nfrom keras.optimizers import Adam\n\ndef make_embed_matrix(embeddings_index, word_index, len_voc):\n    all_embs = np.stack(embeddings_index.values())\n    emb_mean,emb_std = all_embs.mean(), all_embs.std()\n    embed_size = all_embs.shape[1]\n    word_index = word_index\n    embedding_matrix = np.random.normal(emb_mean, emb_std, (len_voc, embed_size))\n    \n    for word, i in word_index.items():\n        if i >= len_voc:\n            continue\n        embedding_vector = embeddings_index.get(word)\n        if embedding_vector is not None: \n            embedding_matrix[i] = embedding_vector\n    \n    return embedding_matrix\n\ndef f1(y_true, y_pred):\n    def recall(y_true, y_pred):\n        true_positives = K.sum(K.round(K.clip(y_true * y_pred, 0, 1)))\n        possible_positives = K.sum(K.round(K.clip(y_true, 0, 1)))\n        recall = true_positives / (possible_positives + K.epsilon())\n        return recall\n\n    def precision(y_true, y_pred):\n        true_positives = K.sum(K.round(K.clip(y_true * y_pred, 0, 1)))\n        predicted_positives = K.sum(K.round(K.clip(y_pred, 0, 1)))\n        precision = true_positives / (predicted_positives + K.epsilon())\n        return precision\n    \n    precision = precision(y_true, y_pred)\n    recall = recall(y_true, y_pred)\n    return 2*((precision*recall)/(precision+recall+K.epsilon()))\n    \ndef make_model(embedding_matrix, embed_size=300):\n    inp = Input(shape=(max_len,))\n    x = Embedding(max_features, embed_size)(inp)\n    x = Bidirectional(CuDNNGRU(64, return_sequences=True))(x)\n    x = GlobalMaxPool1D()(x)\n    x = Dense(16, activation=\"relu\")(x)\n    x = Dropout(0.1)(x)\n    x = Dense(1, activation=\"sigmoid\")(x)\n    \n    model = Model(inputs=inp, outputs=x)\n    model.compile(loss='binary_crossentropy', optimizer='adam', metrics=['accuracy', f1])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"1f411795ca6030063fa46535df45c4c10f5a566b"},"cell_type":"code","source":"print('Creating Models...')\n\nembedding = make_embed_matrix(embed_glove, word_index, len_voc)\nembedding_treated = make_embed_matrix(embed_glove, word_index_treated, len_voc)\n\nmodel = make_model(embedding)\nmodel_treated = make_model(embedding_treated)\n\ndel word_index, word_index_treated\ngc.collect()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"58a9ffebcb2dc552d34bf2541d8c77f0cb80a464"},"cell_type":"code","source":"from keras.callbacks import ModelCheckpoint, ReduceLROnPlateau\n\ncheckpoints = ModelCheckpoint('weights.hdf5', monitor=\"val_f1\", mode=\"max\", verbose=True, save_best_only=True)\ncheckpoints_treated = ModelCheckpoint('treated_weights.hdf5', monitor=\"val_f1\", mode=\"max\", verbose=True, save_best_only=True)\n\nreduce_lr = ReduceLROnPlateau(monitor='val_f1', factor=0.1, patience=2, verbose=1, min_lr=0.000001)\nreduce_lr_treated = ReduceLROnPlateau(monitor='val_f1', factor=0.1, patience=2, verbose=1, min_lr=0.000001)\n\nmodel.summary()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e18c4c47f3ee0744ca150b9dfb4b90007ebf2054"},"cell_type":"code","source":"print('Creating train / test splits...')\nfrom sklearn.model_selection import train_test_split\ny = train['target'].values\nX_train, X_val, y_train, y_val = train_test_split(X, y, test_size=0.1, random_state=420)\nX_t_train, X_t_val, _, _ = train_test_split(X_treated, y, test_size=0.1, random_state=420)\nprint(f\"Training on {X_train.shape[0]} texts\")","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"779d97fb46b3bbe023d886ce7a87b1e2a33750e9"},"cell_type":"markdown","source":"### Fitting The Model"},{"metadata":{"trusted":true,"_uuid":"2170a3463ef001b4ac29c4fc975a3ab3ef3e1261"},"cell_type":"code","source":"epochs = 8\nbatch_size = 512\nhistory = model.fit(X_train, y_train, batch_size=batch_size, epochs=epochs, \n                    validation_data=[X_val, y_val], callbacks=[checkpoints, reduce_lr])\nplt.figure(figsize=(12,8))\nplt.plot(history.history['acc'], label='Train Accuracy')\nplt.plot(history.history['val_acc'], label='Test Accuracy')\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e0b827099a98bec7b44478d8800c996920361a52"},"cell_type":"code","source":"history = model_treated.fit(X_t_train, y_train, batch_size=batch_size, epochs=epochs, \n                            validation_data=[X_t_val, y_val], callbacks=[checkpoints_treated, reduce_lr_treated])\nplt.figure(figsize=(12,8))\nplt.plot(history.history['acc'], label='Train Accuracy')\nplt.plot(history.history['val_acc'], label='Test Accuracy')\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"3ba2f16dabbe088b75ae5d3ec0bc9664e8eea3c9"},"cell_type":"code","source":"model.load_weights('weights.hdf5')\nmodel_treated.load_weights('treated_weights.hdf5')","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"1406764187fc46ae1a5f043765b30617f3cb6cec"},"cell_type":"markdown","source":"### Make Predictions + Calculate Results"},{"metadata":{"trusted":true,"_uuid":"e1c68283494d3c2c7230064e83d0c9944ef4e59f"},"cell_type":"code","source":"from sklearn.metrics import f1_score\n\ndef tweak_threshold(pred, truth):\n    thresholds = []\n    scores = []\n    for thresh in np.arange(0.1, 0.501, 0.01):\n        thresh = np.round(thresh, 2)\n        thresholds.append(thresh)\n        score = f1_score(truth, (pred>thresh).astype(int))\n        scores.append(score)\n    return np.max(scores), thresholds[np.argmax(scores)]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"6fc0f77b2c24f3b840184546f031b885e9fa5a75"},"cell_type":"code","source":"pred_val = model.predict(X_val, batch_size=512, verbose=1)\npred_t_val = model_treated.predict(X_t_val, batch_size=512, verbose=1)\n\nscore_val, threshold_val = tweak_threshold(pred_val, y_val)\nprint(f\"Scored {round(score_val, 4)} for threshold {threshold_val} with untreated texts on validation data\")\n\nscore_t_val, threshold_t_val = tweak_threshold(pred_t_val, y_val)\nprint(f\"Scored {round(score_t_val, 4)} for threshold {threshold_t_val} with treated texts on validation data\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"0fd727451d2c6aa7a6429d08a5f5bccd7b3e6707"},"cell_type":"code","source":"# https://www.kaggle.com/qqgeogor/keras-lstm-attention-glove840b-lb-0-043\nclass Attention(Layer):\n    def __init__(self, step_dim,\n                 W_regularizer=None, b_regularizer=None,\n                 W_constraint=None, b_constraint=None,\n                 bias=True, **kwargs):\n        self.supports_masking = True\n        self.init = initializers.get('glorot_uniform')\n\n        self.W_regularizer = regularizers.get(W_regularizer)\n        self.b_regularizer = regularizers.get(b_regularizer)\n\n        self.W_constraint = constraints.get(W_constraint)\n        self.b_constraint = constraints.get(b_constraint)\n\n        self.bias = bias\n        self.step_dim = step_dim\n        self.features_dim = 0\n        super(Attention, self).__init__(**kwargs)\n\n    def build(self, input_shape):\n        assert len(input_shape) == 3\n\n        self.W = self.add_weight((input_shape[-1],),\n                                 initializer=self.init,\n                                 name='{}_W'.format(self.name),\n                                 regularizer=self.W_regularizer,\n                                 constraint=self.W_constraint)\n        self.features_dim = input_shape[-1]\n\n        if self.bias:\n            self.b = self.add_weight((input_shape[1],),\n                                     initializer='zero',\n                                     name='{}_b'.format(self.name),\n                                     regularizer=self.b_regularizer,\n                                     constraint=self.b_constraint)\n        else:\n            self.b = None\n\n        self.built = True\n\n    def compute_mask(self, input, input_mask=None):\n        return None\n\n    def call(self, x, mask=None):\n        features_dim = self.features_dim\n        step_dim = self.step_dim\n\n        eij = K.reshape(K.dot(K.reshape(x, (-1, features_dim)),\n                        K.reshape(self.W, (features_dim, 1))), (-1, step_dim))\n\n        if self.bias:\n            eij += self.b\n\n        eij = K.tanh(eij)\n\n        a = K.exp(eij)\n\n        if mask is not None:\n            a *= K.cast(mask, K.floatx())\n\n        a /= K.cast(K.sum(a, axis=1, keepdims=True) + K.epsilon(), K.floatx())\n\n        a = K.expand_dims(a)\n        weighted_input = x * a\n        return K.sum(weighted_input, axis=1)\n\n    def compute_output_shape(self, input_shape):\n        return input_shape[0],  self.features_dim","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b6dec9d90f4407d6bc39d1c65ee9a36d743747a3"},"cell_type":"code","source":"inp = Input(shape=(maxlen,))\nx = Embedding(max_features, embed_size, weights=[embedding_matrix])(inp)\nx = Bidirectional(CuDNNGRU(64, return_sequences=True))(x)\nx = Bidirectional(CuDNNGRU(64, return_sequences=True))(x)\nx = Attention(maxlen)(x)\nx = GlobalMaxPool1D()(x)\nx = Dense(16, activation=\"relu\")(x)\nx = Dropout(0.1)(x)\nx = Dense(1, activation=\"sigmoid\")(x)\nmodel = Model(inputs=inp, outputs=x)\nmodel.compile(loss='binary_crossentropy', optimizer='adam', metrics=['accuracy'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f6797ab73bdd5bdb8c8f6d80ec361c50a2b0f56f"},"cell_type":"markdown","source":"\n**References:**\n\nThanks to the below kernels which helped me with this one. \n1. https://www.kaggle.com/jhoward/improved-lstm-baseline-glove-dropout\n2. https://www.kaggle.com/sbongo/do-pretrained-embeddings-give-you-the-extra-edge"}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}