{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport tensorflow as tf\n\nfrom tqdm import tqdm\ntqdm.pandas()\n\nfrom sklearn.model_selection import train_test_split\nfrom sklearn import metrics\n\nfrom keras.preprocessing.text import Tokenizer\nfrom keras.preprocessing.sequence import pad_sequences\n\nimport matplotlib.pyplot as plt\n\nimport time # kernels have a 2 hour limit\n\nfrom collections import Counter","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"6444828abec5a0e0183b06ac88ebb775773ddfaa"},"cell_type":"code","source":"if tf.test.gpu_device_name():\n    print('Default GPU Device: {}'.format(tf.test.gpu_device_name()))\nelse:\n    print(\"Please install GPU version of TF\")","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","collapsed":true,"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":false},"cell_type":"markdown","source":"# Preprocessing\n\nPreprocessing ideas based on https://www.kaggle.com/christofhenkel/how-to-preprocessing-when-using-embeddings. \n\n\nEmbeddings_index code from https://www.kaggle.com/shujian/different-embeddings-with-attention-fork-fork/notebook."},{"metadata":{"trusted":true,"_uuid":"b9685ca6a5bb39ec0116653bc90a87e96f11ef2a"},"cell_type":"code","source":"train = pd.read_csv(\"../input/train.csv\")\ntest = pd.read_csv(\"../input/test.csv\")\nprint(f\"Train shape: {train.shape}\")\nprint(f\"Test shape:  {test.shape}\")\ntrain.sample()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e4c45c89b065d2c6e8ff9eada6ae0cfb73caed3f"},"cell_type":"code","source":"EMBEDDING_FILE = '../input/embeddings/glove.840B.300d/glove.840B.300d.txt' \ndef get_coefs(word,*arr): \n    return word, np.asarray(arr, dtype='float32')  \n\n# creates a mapping from the words to the embedding vectors=\nembeddings_index = dict(get_coefs(*o.split(\" \")) for o in open(EMBEDDING_FILE)) ","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d72506d5f3e44d870d2f804e0224bcfe7cc96966"},"cell_type":"markdown","source":"We want to preprocess our sentences manually to best fit the GloVe embeddings."},{"metadata":{"trusted":true,"_uuid":"c4c6651896608a8f8342b1f931035c8d1fd4cb44"},"cell_type":"code","source":"def check_coverage(vocab,embeddings_index):\n    a, oov, k, i = {}, {}, 0, 0\n    for word in vocab:\n        try:\n            a[word] = embeddings_index[word]\n            k += vocab[word]\n        except:\n            oov[word] = vocab[word]\n            i += vocab[word]\n            pass\n\n    print(f'Found embeddings for {(len(a) / len(vocab)):.2%} of vocab')\n    print(f'Found embeddings for  {(k / (k + i)):.2%} of all text')\n    sorted_x = sorted(oov.items(), key=(lambda x: x[1]), reverse=True)\n\n    return sorted_x\n\ndef get_vocab(question_series):\n    sentences = question_series.str.split().values #get a list of lists of words\n    words = [item for sublist in sentences for item in sublist] # flatten list into just words\n    return dict(Counter(words)) # count words","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"7e71ab9add9dfbbad1421451f061556a0ac18e13"},"cell_type":"code","source":"vocab = get_vocab(train[\"question_text\"])\nout_of_vocab = check_coverage(vocab, embeddings_index)\nout_of_vocab[:10]","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"625ad56ada1039b890c6a402d4a2c1684cc404ae"},"cell_type":"markdown","source":"GloVe has embeddings for certain types of punctuation, so let's keep those in (space seperated) and add an unknown punctuation character."},{"metadata":{"trusted":true,"_uuid":"52e382b8fcf02dc91b91215a1dcd98729b327139"},"cell_type":"code","source":"punct = set('?!.,\"#$%\\'()*+-/:;<=>@[\\\\]^_`{|}~°√' + '“”’')\nembed_punct = punct & set(embeddings_index.keys())\n\ndef clean_punctuation(txt):\n    for p in \"/-\":\n        txt = txt.replace(p, ' ')\n    for p in \"'`‘\":\n        txt = txt.replace(p, '')\n    for p in punct:\n        txt = txt.replace(p, f' {p} ' if p in embed_punct else ' _punct_ ') \n        #known punctuation gets space padded, otherwise we use a newn token\n    return txt","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"65bf91ebe0c21c938ebfc5267ffed02a8081be4e"},"cell_type":"code","source":"train[\"question_text\"] = train[\"question_text\"].map(lambda x: clean_punctuation(x)).str.replace('\\d+', ' # ')\ntest[\"question_text\"] = test[\"question_text\"].map(lambda x: clean_punctuation(x)).str.replace('\\d+', ' # ')\nvocab = get_vocab(train[\"question_text\"])\nout_of_vocab = check_coverage(vocab, embeddings_index)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"0214d477924d8cec0e3c2631944b2cf09e404160"},"cell_type":"code","source":"out_of_vocab[:10]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b8706cd1b42d91cad80827681668a1c85504b0d0"},"cell_type":"code","source":"x = train[\"question_text\"].str.split().map(lambda x: len(x))\nx.describe()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"cb0955e5f96c53e628b6b12968d6f6c1ba638e97"},"cell_type":"code","source":"train, validation = train_test_split(train, test_size=0.08, random_state=20181224)\n\nembed_size = 300 #word vector sizes\nvocab_size = 95000 # words in vocabulary\nmaxlen = 100 # max words to use per question\n\n# fill up the missing values\ntrain_X = train[\"question_text\"].fillna(\"_##_\").values\nval_X = validation[\"question_text\"].fillna(\"_##_\").values\ntest_X = test[\"question_text\"].fillna(\"_##_\").values\n\n# Use Keras to tokenize and pad sequences\ntokenizer = Tokenizer(num_words=vocab_size, filters='', lower=False)\ntokenizer.fit_on_texts(list(train_X))\ntrain_X = tokenizer.texts_to_sequences(train_X)\nval_X = tokenizer.texts_to_sequences(val_X)\ntest_X = tokenizer.texts_to_sequences(test_X)\n\ntrain_X = pad_sequences(train_X, maxlen=maxlen)\nval_X = pad_sequences(val_X, maxlen=maxlen)\ntest_X = pad_sequences(test_X, maxlen=maxlen)\n\n# Get the response\ntrain_y = train['target'].values\nval_y = validation['target'].values","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"ddf9abb83e676185cf9ee43d5a2803a6a77ec874"},"cell_type":"markdown","source":"We will get the mean and standard deviation from the existing embeddings to create random ones for words that do not have embeddings"},{"metadata":{"trusted":true,"_uuid":"44fd5c1d7f78d58425a2905e5951263bd18390f4"},"cell_type":"code","source":"all_embs = np.stack(embeddings_index.values())\nemb_mean,emb_std = all_embs.mean(), all_embs.std()\nembed_size = all_embs.shape[1]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e600ad624600ee2e1f48d3c1ebde7517170aaf9c"},"cell_type":"code","source":"word_index = tokenizer.word_index\nnb_words = min(vocab_size, len(word_index)) # only want at most vocab_size words in our vocabulary \nembedding_matrix = np.random.normal(emb_mean, emb_std, (nb_words, embed_size)) #first, make our embedding matric random7\nnum_missed = 0\nfor word, i in word_index.items(): # insert embeddings we that exist into our matrix\n    if i >= vocab_size: continue\n    embedding_vector = embeddings_index.get(word)\n    if embedding_vector is not None: embedding_matrix[i] = embedding_vector\n    else: num_missed += 1\nprint(num_missed)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"f7a484cad322c41418be622b23ac92ec55176ff6"},"cell_type":"markdown","source":"## Model"},{"metadata":{"trusted":true,"_uuid":"e5f211b06805b1cfbcd3da110450b1debf6bfee6"},"cell_type":"code","source":"BATCH_SIZE = 64\n\n# mmakes sure that all our following operations will be placed in the right graph.\ntf.reset_default_graph()\n\n# should be batchsize x length of each question (vectors of numbers representing indices into the embedding matrix)\nX = tf.placeholder(tf.int32, [None, maxlen], name='X')\n\n# 1d vector with size = None because we want to predict one val for each q, but want variable batch sizes\nY = tf.placeholder(tf.float32, [None], name='Y')\nbatch_size = tf.placeholder(tf.int64, name='batch_size')\n#training = tf.placeholder(tf.bool, name='training')","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"cd19b3957db3cce4edd4fce40f2df6e983666f08"},"cell_type":"markdown","source":"We will use tf.Dataset to load the data for speed and efficiency. We will feed the data using a Reinitializable Iterator so we can shuffle our training data but evaluate our test data without shuffling (more info here: https://towardsdatascience.com/how-to-use-dataset-in-tensorflow-c758ef9e4428). \n\nI do not know an easy way to use tf.Dataset in \"inference mode\", so we will feed it dummy y values for now at inference time."},{"metadata":{"trusted":true,"_uuid":"6e1ffe7fc5fa45064effb70f7c92226f34391e37"},"cell_type":"code","source":"dataset = tf.data.Dataset.from_tensor_slices((X, Y)).shuffle(buffer_size=1000).batch(batch_size).repeat()\ntest_dataset = tf.data.Dataset.from_tensor_slices((X, Y)).batch(batch_size) #this one does not shuffle\n\niterator = tf.data.Iterator.from_structure(dataset.output_types,\n                                           dataset.output_shapes) \n\n# To choose which dataset we use, we simply initialize the appropriate one using the init_ops below\ntrain_init_op = iterator.make_initializer(dataset)\ntest_init_op = iterator.make_initializer(test_dataset)\n\nquestions, labels = iterator.get_next()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"58d830282223ef7bbc39d70d8905ffaf841fb3d5"},"cell_type":"markdown","source":"Our embeddings layer will not be trainable for now, perhaps in a future version we can make them trainable part way into training. If we make them trainable too early, the embeddings will be destroyed due to the random weights of the actual model."},{"metadata":{"trusted":true,"_uuid":"faf6b146fe0d5f680145907217b226e472eb5211"},"cell_type":"code","source":"embeddings = tf.get_variable(name=\"embeddings\", shape=embedding_matrix.shape,\n                             initializer=tf.constant_initializer(np.array(embedding_matrix)), \n                             trainable=False)\nembed = tf.nn.embedding_lookup(embeddings, questions)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d002b54119451da120f458d598435cf686c6db33"},"cell_type":"markdown","source":"The self-attention mechanism below is based on work of Vaswani et al. (cited on this paper http://www.aclweb.org/anthology/W18-3002).\n\nThe paper describes using L2 regularization; however, I found using no regularization gave me the best validation result. "},{"metadata":{"trusted":true,"_uuid":"9efb0893c9fd7194b71398f2c7a1730da1f0d32f"},"cell_type":"code","source":"l2_amt = 0.0\nregularizer = tf.contrib.layers.l2_regularizer(l2_amt)\nQ = tf.nn.elu(tf.layers.conv1d(embed, embed_size, 3, padding='SAME', kernel_regularizer=regularizer))\nK = tf.nn.elu(tf.layers.conv1d(embed, embed_size, 3, padding='SAME', kernel_regularizer=regularizer))\nV = tf.nn.elu(tf.layers.conv1d(embed, embed_size, 3, padding='SAME', kernel_regularizer=regularizer))\nQKT = tf.matmul(Q, K, transpose_b=True)/np.sqrt(embed_size)\n\n#softmax manual computation\nexp_QKT = tf.exp(QKT)\nqkt_sum = tf.reduce_sum(exp_QKT, axis=[1,2])\nsoftmax = exp_QKT/tf.expand_dims(tf.expand_dims(qkt_sum, 1), 1)\n\nattention = tf.matmul(softmax, V)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"b0d4142468a2e72587bd65ab55af256e43336f63"},"cell_type":"markdown","source":"The architecture below is based on work by Zhang & Wallace, 2016 (https://arxiv.org/pdf/1510.03820.pdf). Previously, I tried a 2D CNN and a deeper 1D cnn but this approach seemed to work best. \n\nThe paper recommended feature maps of sizes 100-600 with 3 convolution sizes. I used a small feature map with more convolution sizes and got a good result."},{"metadata":{"trusted":true,"_uuid":"54d6726be838ebf622cb6a8abc479d8180e5dfa1"},"cell_type":"code","source":"kernel_lens = [3, 4, 5, 6, 7] # (3, 4, 5) and (7, 7, 7) worked well in the paper\nnum_filters = len(kernel_lens)\n\nZ = [tf.layers.conv1d(attention, 100, ker_len, padding='SAME', kernel_regularizer=regularizer) \n     for ker_len in kernel_lens]\n#Z = [tf.nn.conv1d(embed, W[i], stride = 1, padding = 'SAME') for i in range (num_filters) ]\nA = [tf.nn.relu(Z[i]) for i in range (num_filters)]\nP = [tf.reduce_max(A[i], axis=2) for i in range (num_filters)]\n\nFLAT = tf.contrib.layers.flatten(tf.concat(P, axis=1))\n\nlast_layer = tf.layers.dense(FLAT, 1, kernel_regularizer=regularizer) #fully connected layer\nprediction = tf.nn.sigmoid(last_layer) #activation function\nprediction = tf.squeeze(prediction, [1]) # layers.dense returns a tensor, but we want to remove the extra dimension","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"970a80415d4c1534b407cba577d78a661b45c6b4"},"cell_type":"markdown","source":"Adam and MomentumOptimizer were tested along with various learning rates from 0.001 to 0.01. Momentum with a high learning rate worked best."},{"metadata":{"trusted":true,"_uuid":"85c408559debe74847f9820c605ffa85dbc47083"},"cell_type":"code","source":"learning_rate=0.01\n\nl2_loss = tf.losses.get_regularization_loss()\n\n# define cross entropy loss function\nloss = tf.nn.sigmoid_cross_entropy_with_logits(logits=tf.squeeze(last_layer), labels=labels)\nloss = tf.reduce_mean(loss) + l2_loss\n\n# define our optimizer to minimize the loss\noptimizer = tf.train.MomentumOptimizer(learning_rate, 0.9).minimize(loss) #MomentumOptimizer","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"fb81cac4d837e6bc6d422aca052eec81828cbc30"},"cell_type":"markdown","source":"F1 is our target metric, so we will track that. Every minibatch, you run f1_update, which keeps track of the true pos/false pos/false negs. When you run f1, it will compute an optimal F1 score (based on 200 threshholds). You run the reset_op to reset these counters for each batch."},{"metadata":{"trusted":true,"_uuid":"4bdcce2a0778cbb7ee105ee20c2c9c0d3b3968bd"},"cell_type":"code","source":"with tf.name_scope('metrics'):\n    F1, f1_update = tf.contrib.metrics.f1_score(labels=labels, predictions=prediction, name='my_metric')\n    \nrunning_vars = tf.get_collection(tf.GraphKeys.LOCAL_VARIABLES, scope=\"my_metric\")\nreset_op = tf.variables_initializer(var_list=running_vars)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"2545ae80248d7eea922214f1aad856590f1338c2"},"cell_type":"markdown","source":"## Training"},{"metadata":{"trusted":true,"_uuid":"9942110307254cecef3d5c36fe72e924f022a1c9"},"cell_type":"code","source":"num_epochs = 10\nseed = 3 # we use a seed to have deterministic results\n\nsess = tf.Session()\n\n# Run the initialization\nsess.run(tf.global_variables_initializer()) # initializes all of our variables\nsess.run(tf.local_variables_initializer()) # need this for the f1 metric to work\n\ncosts, f1, val_f1 = [], [], []","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"9acd8b4b0778cb9b22cf705b48471f6fa610ad7e"},"cell_type":"code","source":"def check_validation(sz = 90, plot=False):\n    sess.run(test_init_op, feed_dict={X: val_X, Y: val_y, batch_size: sz})\n    val_pred = np.concatenate([sess.run(prediction) for _ in range(int(val_X.shape[0]/sz))])\n    thresholds = [i/200 for i in range(10, 120, 1)] \n    scores = [metrics.f1_score(val_y,np.int16(val_pred > t)) for t in thresholds]\n    \n    if plot:\n        plt.plot(thresholds, scores)\n        plt.ylabel(\"F1 Score\")\n        plt.xlabel('Threshold')\n        plt.title(\"F1 Score by thresholds for Validation Set\")\n        plt.show()\n        \n    return max(scores), thresholds[np.argmax(scores)]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"46e86d664b34c4aa5afa041e6bc2e82d63fee66d","scrolled":true},"cell_type":"code","source":"## Want to train for 6600 seconds to stay under the time limit of 7200 seconds (2 hours)\nstart = time.time()\nend = 0\nmax_time = 6700\n\n# initialize iterator with train data\nsess.run(train_init_op, feed_dict={X:train_X, Y:train_y, batch_size:BATCH_SIZE})\nnum_iter = 2000 # print after each num_iter\nnum_batches = int(train_X.shape[0] / BATCH_SIZE) # number of batches/minibatches\n\n# Training Loop\nfor epoch in range(1, num_epochs+1):\n    seed += seed # want a different random shuffle every time, but still have deterministic results\n    tf.set_random_seed(seed)\n    iter_cost = 0.\n    \n    # the last batch is smaller than the rest, so we will use \n    # this to keep track of the number of iterations to get the right average cost\n    prev_iter = 0. \n    \n    for i in range(num_batches):\n        _ , batch_loss, _ = sess.run([optimizer, loss, f1_update]) \n        iter_cost += batch_loss\n        \n        # End training after \n        end = time.time()\n        if (end-start > max_time): \n            break\n        \n        if (i % num_iter == 0 and i > 0): \n            iter_cost /= (i-prev_iter) # get average batch cost\n            prev_iter = i #update prev_iter for next iteration\n            cur_f1 = sess.run(F1)\n            sess.run(reset_op) # reset counters for F1\n            \n            f1.append(cur_f1)\n            costs.append(iter_cost)\n            print (f\"Epoch {epoch} Iteration {i:5}  cost: {iter_cost:.6f}  f1: {cur_f1:.5f}  time: {end-start:4.4f}\")\n            batch_cost = 0. #reset batch_cost\n            \n    val_f1.append(check_validation()[0])\n    print(f'val f1 {val_f1[-1]}')\n    sess.run(train_init_op, feed_dict={X:train_X, Y:train_y, batch_size:BATCH_SIZE})\n    if (end-start > max_time): \n        break","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"3e2bdd2e6e978e2ac7abd53e646c4c1c49f6abf8"},"cell_type":"markdown","source":"## Evaluate Model with Validation Set"},{"metadata":{"trusted":true,"_uuid":"c6aa61b1b9c3884b6be18b8409899af75506d8ec"},"cell_type":"code","source":"def easy_plot(yvals, ylabel='', num_iter=num_iter):\n    plt.plot(yvals)\n    plt.ylabel(ylabel)\n    plt.xlabel(f'Iterations (per {num_iter})')\n    plt.title(f\"{ylabel} by Iterations for Learning Rate = {learning_rate}\")\n    plt.show()\n    \neasy_plot(np.squeeze(costs), 'Cost')\neasy_plot(np.squeeze(f1), 'Train F1 Score')\neasy_plot(np.squeeze(val_f1), 'Validation F1 Score', num_iter=20000)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"692550c24c2322e116411ba2211baa41f4bf0cc3"},"cell_type":"code","source":"tf.set_random_seed(2018)\nscore, thresh = check_validation(plot=True)\nprint(f\"Best Validation F1 Score is {score:.4f} at threshold {thresh}\")","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"19342380fb073f5641de535cbd61dc4c138e4947"},"cell_type":"markdown","source":"## Predict on Test Set"},{"metadata":{"trusted":true,"_uuid":"bf82f67c2c2dd8f47136853842e9fb144245a01c"},"cell_type":"code","source":"sz=30\ntemp_y = val_y[:test_X.shape[0]]\nsub = test[['qid']]\nsess.run(test_init_op, feed_dict={X: test_X, Y: temp_y, batch_size:sz})\nsub['prediction'] = np.concatenate([sess.run(prediction) for _ in range(int(test_X.shape[0]/sz))])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"3b75504411f98abdd41504dc86c53e952ce718d2"},"cell_type":"code","source":"sub['prediction'] = (sub['prediction'] > thresh).astype(np.int16)\nsub.to_csv(\"submission.csv\", index=False)\nsub.sample()","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}