{"cells":[{"metadata":{"_uuid":"f38a71fc04b07f09e4eb601d0e2aca90f63eaac7"},"cell_type":"markdown","source":"Using keras\nand glove embeddings"},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"#imports\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nfrom sklearn.model_selection import train_test_split\nfrom sklearn import metrics\n\nfrom keras.models import Model\nfrom keras.layers import Input, Embedding, Bidirectional, Dense, CuDNNLSTM, CuDNNGRU, Dropout, SpatialDropout1D, Concatenate, BatchNormalization\nfrom keras.preprocessing.sequence import pad_sequences\nfrom keras.preprocessing.text import Tokenizer\nfrom keras import backend as K\nfrom keras.engine.topology import Layer\nfrom keras import initializers, regularizers, constraints, optimizers\n\nimport os\nprint(os.listdir(\"../input\"))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"8146b9b91393aeb9388ab0adbef3e5f3675fa61a"},"cell_type":"code","source":"def load_data():\n    train_df = pd.read_csv(\"../input/train.csv\")\n    test_df = pd.read_csv(\"../input/test.csv\")\n    print(\"Train shape : \",train_df.shape)\n    print(\"Test shape : \",test_df.shape)\n    return train_df, test_df","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d9fb86c5c56e916a8c41fe8ba5e0d0b129ba36a7"},"cell_type":"markdown","source":"**Preprocessing**\n\n Preprocessing ideas based on https://www.kaggle.com/christofhenkel/how-to-preprocessing-when-using-embeddings."},{"metadata":{"trusted":true,"scrolled":true,"_uuid":"7c051c658e0e342d6f29b71e4317f5c2cf7747ba"},"cell_type":"code","source":"train_df, test_df = load_data()\ntrain_df.sample()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b4c6d7b881685c6fce863287da453ea24b7b8f33"},"cell_type":"code","source":"!ls ../input/embeddings/","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"1514c0e31a08b8a6151f5904c6a7c6492b856397"},"cell_type":"markdown","source":"Glove Embeddings"},{"metadata":{"trusted":true,"_uuid":"b9f6389c01c764e892f5cceb0934c60619466aa1"},"cell_type":"code","source":"def get_coefs(word,*arr): return word, np.asarray(arr, dtype='float32')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"scrolled":true,"_uuid":"baf2b3a588e18fefc649991f4d84441372cbab29"},"cell_type":"code","source":"EMBEDDING_FILE = '../input/embeddings/glove.840B.300d/glove.840B.300d.txt'\n\nembeddings_index = dict(get_coefs(*o.split(\" \")) for o in open(EMBEDDING_FILE))\nprint('Found %s word vectors.' % len(embeddings_index))","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"7422ab7bc16ba234a05fecd6d942ca2a98013d49"},"cell_type":"markdown","source":"Preprocess sentence to best fit glove embeddings"},{"metadata":{"trusted":true,"_uuid":"2f39e2355a0d0569b66cfea8b5e75ecd9de257d0"},"cell_type":"code","source":"from collections import Counter\n\ndef check_coverage(vocab,embeddings_index):\n    a, oov, k, i = {}, {}, 0, 0\n    for word in vocab:\n        try:\n            a[word] = embeddings_index[word]\n            k += vocab[word]\n        except:\n            oov[word] = vocab[word]\n            i += vocab[word]\n            pass\n\n    print(f'Found embeddings for {(len(a) / len(vocab)):.2%} of vocab')\n    print(f'Found embeddings for  {(k / (k + i)):.2%} of all text')\n    sorted_x = sorted(oov.items(), key=(lambda x: x[1]), reverse=True)\n\n    return sorted_x\n\ndef get_vocab(question_series):\n    sentences = question_series.str.split().values #get a list of lists of words\n    words = [item for sublist in sentences for item in sublist] # flatten list into just words\n    return dict(Counter(words)) # count words","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d974c1674b716361d340afbccb86b3d8d2409e14"},"cell_type":"code","source":"vocab = get_vocab(train_df[\"question_text\"])\nout_of_vocab = check_coverage(vocab, embeddings_index)\nout_of_vocab[:10]","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"aceaa2ade13f057333d63e8318cb2171b0bd1371"},"cell_type":"markdown","source":"GloVe has embeddings for certain types of punctuation, so let's keep those in (space seperated) and add an unknown punctuation character."},{"metadata":{"trusted":true,"_uuid":"e02f78001cdaa252f0d95607b604a5f785f3815d"},"cell_type":"code","source":"punct = set('?!.,\"#$%\\'()*+-/:;<=>@[\\\\]^_`{|}~°√' + '“”’')\nembed_punct = punct & set(embeddings_index.keys())\n\ndef clean_punctuation(txt):\n    for p in \"/-\":\n        txt = txt.replace(p, ' ')\n    for p in \"'`‘\":\n        txt = txt.replace(p, '')\n    for p in punct:\n        txt = txt.replace(p, f' {p} ' if p in embed_punct else ' _punct_ ') \n        #known punctuation gets space padded, otherwise we use a newn token\n    return txt","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e9c7b98b83c23c2564c879d57e9bdbed641d1baf"},"cell_type":"code","source":"train_df[\"question_text\"] = train_df[\"question_text\"].map(lambda x: clean_punctuation(x)).str.replace('\\d+', ' # ')\ntest_df[\"question_text\"] = test_df[\"question_text\"].map(lambda x: clean_punctuation(x)).str.replace('\\d+', ' # ')\nvocab = get_vocab(train_df[\"question_text\"])\nout_of_vocab = check_coverage(vocab, embeddings_index)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"scrolled":true,"_uuid":"296eca517edaa9e1f3682772cd4eaf73a328669c"},"cell_type":"code","source":"len(vocab) - len(out_of_vocab)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"207d6d993e5442a3ba05dbb5fa048fadd9703bf4"},"cell_type":"markdown","source":"Code taken from  kernel\n\nNext steps are as follows:\n\nSplit the training dataset into train and val sample. Cross validation is a time consuming process and so let us do simple train val split.\nFill up the missing values in the text column with 'na'\nTokenize the text column and convert them to vector sequences\nPad the sequence as needed - if the number of words in the text is greater than 'max_len' trunacate them to 'max_len' or if the number of words in the text is lesser than 'max_len' add zeros for remaining values."},{"metadata":{"trusted":true,"_uuid":"8611d0addac88a93d5ad40f99cef86331e144c61"},"cell_type":"code","source":"maxlen = 65 # max number of words in a question to use\nmax_features = 60000 # how many unique words to use (i.e num rows in embedding vector)\n\n## split to train and val\ntrain_df, val_df = train_test_split(train_df, test_size=0.1, random_state=201901)\n\n# fill up the missing values\ntrain_X = train_df[\"question_text\"].fillna(\"_##_\").values\nval_X = val_df[\"question_text\"].fillna(\"_##_\").values\ntest_X = test_df[\"question_text\"].fillna(\"_##_\").values\n\n## Tokenize the sentences\ntokenizer = Tokenizer(num_words=max_features)\ntokenizer.fit_on_texts(list(train_X))\ntrain_X = tokenizer.texts_to_sequences(train_X)\nval_X = tokenizer.texts_to_sequences(val_X)\ntest_X = tokenizer.texts_to_sequences(test_X)\n\n## Pad the sentences \ntrain_X = pad_sequences(train_X, maxlen=maxlen)\nval_X = pad_sequences(val_X, maxlen=maxlen)\ntest_X = pad_sequences(test_X, maxlen=maxlen)\n\n## Get the target values\ntrain_y = train_df['target'].values\nval_y = val_df['target'].values","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"65743f05224a38673468b13e927695ef18eca4ca"},"cell_type":"code","source":"def prepare_embedding_matrix(embeddings_index,word_index,num_words):\n    all_embs = np.stack(embeddings_index.values())\n    emb_mean,emb_std = all_embs.mean(), all_embs.std()\n    embed_size = all_embs.shape[1]\n\n    embedding_matrix = np.random.normal(emb_mean, emb_std, (num_words, embed_size))\n    #embedding_matrix = np.zeros((num_words, EMBEDDING_DIM))\n    for word, i in word_index.items():\n        if i >= max_features:\n            continue\n        embedding_vector = embeddings_index.get(word)\n        if embedding_vector is not None:\n            embedding_matrix[i] = embedding_vector\n    return embedding_matrix","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"20facc407b8aecb078b6ac6e335a7300bc14c08c"},"cell_type":"code","source":"EMBEDDING_DIM = 300 # how big is each word vector\nword_index = tokenizer.word_index\nnum_words = min(max_features, len(word_index) + 1)\nembedding_matrix = prepare_embedding_matrix(embeddings_index,word_index,num_words)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"6ed5b71f7f66e8ec61c3982e93c76a528e5301ab"},"cell_type":"markdown","source":"**Attention Layer:** [https://www.kaggle.com/suicaokhoailang/lstm-attention-baseline-0-652-lb](https://www.kaggle.com/suicaokhoailang/lstm-attention-baseline-0-652-lb)"},{"metadata":{"trusted":true,"_uuid":"d8b79dc1088465066c6a300b1d922dd74d2b6fae"},"cell_type":"code","source":"class Attention(Layer):\n    def __init__(self, step_dim,\n                 W_regularizer=None, b_regularizer=None,\n                 W_constraint=None, b_constraint=None,\n                 bias=True, **kwargs):\n        self.supports_masking = True\n        self.init = initializers.get('glorot_uniform')\n\n        self.W_regularizer = regularizers.get(W_regularizer)\n        self.b_regularizer = regularizers.get(b_regularizer)\n\n        self.W_constraint = constraints.get(W_constraint)\n        self.b_constraint = constraints.get(b_constraint)\n\n        self.bias = bias\n        self.step_dim = step_dim\n        self.features_dim = 0\n        super(Attention, self).__init__(**kwargs)\n\n    def build(self, input_shape):\n        assert len(input_shape) == 3\n\n        self.W = self.add_weight((input_shape[-1],),\n                                 initializer=self.init,\n                                 name='{}_W'.format(self.name),\n                                 regularizer=self.W_regularizer,\n                                 constraint=self.W_constraint)\n        self.features_dim = input_shape[-1]\n\n        if self.bias:\n            self.b = self.add_weight((input_shape[1],),\n                                     initializer='zero',\n                                     name='{}_b'.format(self.name),\n                                     regularizer=self.b_regularizer,\n                                     constraint=self.b_constraint)\n        else:\n            self.b = None\n\n        self.built = True\n\n    def compute_mask(self, input, input_mask=None):\n        return None\n\n    def call(self, x, mask=None):\n        features_dim = self.features_dim\n        step_dim = self.step_dim\n\n        eij = K.reshape(K.dot(K.reshape(x, (-1, features_dim)),\n                        K.reshape(self.W, (features_dim, 1))), (-1, step_dim))\n\n        if self.bias:\n            eij += self.b\n\n        eij = K.tanh(eij)\n\n        a = K.exp(eij)\n\n        if mask is not None:\n            a *= K.cast(mask, K.floatx())\n\n        a /= K.cast(K.sum(a, axis=1, keepdims=True) + K.epsilon(), K.floatx())\n\n        a = K.expand_dims(a)\n        weighted_input = x * a\n        return K.sum(weighted_input, axis=1)\n\n    def compute_output_shape(self, input_shape):\n        return input_shape[0],  self.features_dim","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"6b3451fc37c2ebecfb1f2076f9bcf7fbca483aca"},"cell_type":"code","source":"inp = Input(shape=(maxlen,))\nx = Embedding(max_features, EMBEDDING_DIM, weights=[embedding_matrix],trainable=False)(inp)\nx = SpatialDropout1D(0.25)(x)\nx1 = Bidirectional(CuDNNLSTM(128, return_sequences=True))(x)\nx2 = Bidirectional(CuDNNGRU(128, return_sequences=True))(x)\nattn_lstm = Attention(maxlen)(x1)\nattn_gru = Attention(maxlen)(x2)\nconcat = Concatenate()([attn_lstm,attn_gru])\nconcat = BatchNormalization()(concat)\nd = Dense(256, activation=\"relu\")(concat)\nd = Dropout(0.3)(d)\nout = Dense(1, activation=\"sigmoid\")(d)\nmodel = Model(inputs=inp, outputs=out)\nmodel.compile(loss='binary_crossentropy', optimizer='adam', metrics=['accuracy'])\nprint(model.summary())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"bdabcfb2604f2b687ca1b0d54d84c10a06cee883"},"cell_type":"code","source":"model.fit(train_X, train_y, batch_size=512, epochs=4, validation_data=(val_X, val_y))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"82fdcfd9bcf24f8106991586593e610507ca7287"},"cell_type":"code","source":"pred_glove_val_y = model.predict([val_X], batch_size=1024, verbose=1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"9e0f75aaece0be440e618cc666ec564c54ae0098"},"cell_type":"code","source":"from sklearn.metrics import roc_curve, precision_recall_curve\ndef threshold_search(y_true, y_proba, plot=False):\n    precision, recall, thresholds = precision_recall_curve(y_true, y_proba)\n    thresholds = np.append(thresholds, 1.001) \n    F = 2 / (1/precision + 1/recall)\n    best_score = np.max(F)\n    best_th = thresholds[np.argmax(F)]\n    if plot:\n        plt.plot(thresholds, F, '-b')\n        plt.plot([best_th], [best_score], '*r')\n        plt.show()\n    search_result = {'threshold': best_th , 'f1': best_score}\n    return search_result ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"32c5ff4d94cb73b8d4be24fb2dc01a885731d19e"},"cell_type":"code","source":"result = threshold_search(val_y, pred_glove_val_y)\nprint(result)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ecb3aab07e911f0fdad4c498f9311be1e4493d78"},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d4829658af258fa86db5c0855045cb3342543e25"},"cell_type":"code","source":"from keras.models import load_model\nmodel.save('my_model.h5')  # creates a HDF5 file 'my_model.h5'\n#del model  # deletes the existing model\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ac77dc7791fa844ac65d4a8e96cff798d0106ffb"},"cell_type":"code","source":"\n# returns a compiled model\n# identical to the previous one\n#model = load_model('/out/my_model.h5')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"38640296d8040f2d337cae83289fe5669696030d"},"cell_type":"code","source":"pred_glove_test_y = model.predict([test_X], batch_size=1024, verbose=1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"1337679ae45cb1036b6184814ac8cf86ea1002d7"},"cell_type":"code","source":"\npred_test_y = pred_glove_test_y\npred_test_y = (pred_test_y > result['threshold']).astype(int)\nout_df = pd.DataFrame({\"qid\":test_df[\"qid\"].values})\nout_df['prediction'] = pred_test_y\nout_df.to_csv(\"submission.csv\", index=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d8fdfdaf3b61febc9d8d49be93df77861c89de32"},"cell_type":"code","source":"out_df.describe()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"0057e7fbe819851fdf69cc37d441d075f4632e87"},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}