{"cells":[{"metadata":{"trusted":true,"_uuid":"6408784734669f8bc5e5b65c1608d2f57639d5c6"},"cell_type":"markdown","source":"And the credit goes to [Miha Skalic](http://www.kaggle.com/mihaskalic) for [LSTM is all you need! well, maybe embeddings also](https://www.kaggle.com/mihaskalic/lstm-is-all-you-need-well-maybe-embeddings-also). This is a modied version of it.  "},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"from keras.preprocessing.text import Tokenizer\nfrom keras.preprocessing.sequence import pad_sequences\n\nimport os\nimport gc\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nfrom tqdm import tqdm\nimport math\nfrom sklearn.model_selection import train_test_split\n\nfrom nltk.corpus import stopwords\neng_stopwords = set(stopwords.words(\"english\"))\nimport string","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"21f85dfd5b9bf3d8b27ba29149d52253e5d64049"},"cell_type":"markdown","source":"# Setup"},{"metadata":{"trusted":true,"_uuid":"78578eab64a477d0a5ad6b1c917ae154868a44df"},"cell_type":"code","source":"train_df = pd.read_csv(\"../input/train.csv\")\ngc.collect()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"7d38d5b95c5030fc152cabd66f877a499034c230"},"cell_type":"code","source":"test_df = pd.read_csv(\"../input/test.csv\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"4c31dfdb5633fa347f20fe784be9f45790d659f2"},"cell_type":"code","source":"train_df[\"question_text\"].isna().sum(), test_df[\"question_text\"].isna().sum(), ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"4c344c6036bc39185373712e693788f30d71effa"},"cell_type":"code","source":"train_df, val_df = train_test_split(train_df, test_size=0.1, random_state = 1001)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f5a6d99de96d4b7e01c88740cc03a0a7a63dd777"},"cell_type":"code","source":"train_df.shape, val_df.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"261a4b3ec308d7a39f0f495ee9447d0fef6a8cad"},"cell_type":"code","source":"max_features = 95000\nmax_len = 72\nembed_size = 300","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"92ffbf2ef35d2dc5863ee27c61eadd1869ca5440"},"cell_type":"code","source":"%%time\n# embdedding setup\n# Source https://blog.keras.io/using-pre-trained-word-embeddings-in-a-keras-model.html\nembeddings_index = {}\nf = open('../input/embeddings/glove.840B.300d/glove.840B.300d.txt')\nfor line in tqdm(f):\n    values = line.split(\" \")\n    word = values[0]\n    coefs = np.asarray(values[1:], dtype='float32')\n    embeddings_index[word] = coefs\nf.close()\ngc.collect()\nprint('Found %s word vectors.' % len(embeddings_index))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"9cc8b0bcd285225ce651d024f506615d4656b9f7"},"cell_type":"code","source":"# Convert values to embeddings\ndef text_to_array(text, zeros = 300, split_val = max_len):\n    empyt_emb = np.zeros(zeros)\n    text = text[:-1].split()[:split_val]\n    embeds = [embeddings_index.get(x, empyt_emb) for x in text]\n    embeds+= [empyt_emb] * (split_val - len(embeds))\n    return np.array(embeds)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"9cc8b0bcd285225ce651d024f506615d4656b9f7"},"cell_type":"code","source":"# train_vects = [text_to_array(X_text) for X_text in tqdm(train_df[\"question_text\"])]\nval_vects = np.array([text_to_array(X_text) for X_text in tqdm(val_df[\"question_text\"][:5000])]) \nval_y = np.array(val_df[\"target\"][:5000])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"0c950448e0717eaebb920d93cfdc6b4561e21853"},"cell_type":"code","source":"# Data providers\nbatch_size = 256\n\ndef batch_gen(train_df):\n    n_batches = math.ceil(len(train_df) / batch_size)\n    while True: \n        train_df = train_df.sample(frac=1.)  # Shuffle the data.\n        for i in range(n_batches):\n            texts = train_df.iloc[i*batch_size:(i+1)*batch_size, 1]\n            text_arr = np.array([text_to_array(text) for text in texts])\n            yield text_arr, np.array(train_df[\"target\"][i*batch_size:(i+1)*batch_size])\n","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"706c0224b6112e5a8f00ad25f35e90fbb9519a5f"},"cell_type":"markdown","source":"# Training"},{"metadata":{"trusted":true,"_uuid":"de1448acbaf40b8bd7197533704cabc90c2af9f4"},"cell_type":"code","source":"from keras.models import Sequential,Model\nfrom keras.layers import CuDNNLSTM, Dense, Bidirectional, Input,Dropout, CuDNNGRU\nfrom keras.engine.topology import Layer\nfrom keras import initializers, regularizers, constraints\nfrom keras.callbacks import Callback, EarlyStopping, ModelCheckpoint\nfrom sklearn.metrics import f1_score\nfrom keras.layers.normalization import BatchNormalization\nfrom keras import backend as K\nfrom keras.engine.topology import Layer, InputSpec","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ba5bb8eedc44f972d27ab885217aa497235f46b4"},"cell_type":"code","source":"class Attention(Layer):\n    def __init__(self, step_dim,\n                 W_regularizer=None, b_regularizer=None,\n                 W_constraint=None, b_constraint=None,\n                 bias=True, **kwargs):\n        self.supports_masking = True\n        self.init = initializers.get('normal')\n\n        self.W_regularizer = regularizers.get(W_regularizer)\n        self.b_regularizer = regularizers.get(b_regularizer)\n\n        self.W_constraint = constraints.get(W_constraint)\n        self.b_constraint = constraints.get(b_constraint)\n\n        self.bias = bias\n        self.step_dim = step_dim\n        self.features_dim = 0\n        super(Attention, self).__init__(**kwargs)\n\n    def build(self, input_shape):\n        assert len(input_shape) == 3\n\n        self.W = self.add_weight((input_shape[-1],),\n                                 initializer=self.init,\n                                 name='{}_W'.format(self.name),\n                                 regularizer=self.W_regularizer,\n                                 constraint=self.W_constraint)\n        self.features_dim = input_shape[-1]\n\n        if self.bias:\n            self.b = self.add_weight((input_shape[1],),\n                                     initializer='zero',\n                                     name='{}_b'.format(self.name),\n                                     regularizer=self.b_regularizer,\n                                     constraint=self.b_constraint)\n        else:\n            self.b = None\n\n        self.built = True\n\n    def compute_mask(self, input, input_mask=None):\n        return None\n\n    def call(self, x, mask=None):\n        features_dim = self.features_dim\n        step_dim = self.step_dim\n\n        eij = K.reshape(K.dot(K.reshape(x, (-1, features_dim)),\n                        K.reshape(self.W, (features_dim, 1))), (-1, step_dim))\n\n        if self.bias:\n            eij += self.b\n\n        eij = K.tanh(eij)\n\n        a = K.exp(eij)\n\n        if mask is not None:\n            a *= K.cast(mask, K.floatx())\n\n        a /= K.cast(K.sum(a, axis=1, keepdims=True) + K.epsilon(), K.floatx())\n\n        a = K.expand_dims(a)\n        weighted_input = x * a\n        return K.sum(weighted_input, axis=1)\n\n    def compute_output_shape(self, input_shape):\n        return input_shape[0],  self.features_dim","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"6342db0bf0dda2a7c76b9c5987bd9546dd945740"},"cell_type":"code","source":"class F1Evaluation(Callback):\n    def __init__(self, validation_data=(), interval=1):\n        super(Callback, self).__init__()\n        self.interval = interval\n        self.X_val, self.y_val = validation_data\n\n    def on_epoch_end(self, epoch, logs={}):\n        if epoch % self.interval == 0:\n            y_pred = self.model.predict(self.X_val, verbose=0)\n            y_pred = (y_pred > 0.5).astype(int)\n            score = f1_score(self.y_val, y_pred)\n            print(\"\\n F1 - Epoch: %d - Score: %.6f \\n\" % (epoch+1, score)) ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c265bbb18cc2a9bdc2febe81a2db6150ffca0c22"},"cell_type":"code","source":"esr = EarlyStopping(verbose=2, patience=3)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c9ee258275be55cc0404f00237379494c7f8745e"},"cell_type":"code","source":"f1 = F1Evaluation(validation_data=(val_vects, val_y), interval=1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e74e41c714ec637580c5c22d7c96c86497bfbf75"},"cell_type":"code","source":"inp = Input(shape=(max_len,300 ))\nx = Bidirectional(CuDNNGRU(192, return_sequences=True))(inp)\nx = Bidirectional(CuDNNGRU(64,return_sequences=True))(x)\nx = Attention(max_len)(x)\n#x = Dropout(0.25)(x)\nx = Dense(32, activation=\"relu\")(x)\nx = Dropout(0.25)(x)\nx = Dense(1, activation=\"sigmoid\")(x)\nmodel = Model(inputs=inp, outputs=x)\nmodel.compile(loss='binary_crossentropy',\n              optimizer='adam',\n              metrics=['accuracy'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"afe8afeae79ddd037f7648fce71030c1c34dad46"},"cell_type":"code","source":"model.summary()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"400a3e45972fb74cef2d94d8d3d87136597e5f72"},"cell_type":"code","source":"from keras.utils import plot_model\nplot_model(model, to_file='model.png',show_shapes=True)\n\nfrom IPython.display import Image\nImage(filename='model.png')\n#from keras.utils.vis_utils import model_to_dot\n\n#SVG(model_to_dot(model).create(prog='dot', format='svg'))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"9eea551e09a9243d1bea15fdef4a73d8c2d3f76d"},"cell_type":"code","source":"model_name = 'gru_model'#%(rate_drop_lstm,rate_drop_dense)\nprint(model_name)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"6c9038b7596bd1fdfa665f20c2e22ce7b5a049f8"},"cell_type":"code","source":"bst_model_path = model_name + '.h5'\nmodel_checkpoint = ModelCheckpoint(bst_model_path, save_best_only=True, save_weights_only=True, verbose=1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f35d1f82f90825ca0d032b3002010d70569c0ddd"},"cell_type":"code","source":"gc.collect() ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"3f0b703ded691293450beb4ddf2d903d0e08c737","scrolled":true},"cell_type":"code","source":"np.random.seed(2018)\nmg = batch_gen(train_df)\nhist = model.fit_generator(mg, epochs=30,\n                    steps_per_epoch=512,\n                    validation_data=(val_vects, val_y), callbacks=[f1, esr, model_checkpoint],\n                    verbose=2)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"3b08c14a424fda6b7361d1214fd3c3e41bbef770"},"cell_type":"code","source":"model.load_weights(bst_model_path)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ced034968d0477a767589e31c4409a1a4096c2f3"},"cell_type":"code","source":"import matplotlib.pyplot as plt ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"0f9b767e52e919f71b97e835c12a899c3927944b"},"cell_type":"code","source":"gc.collect()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f36cb092b0b76b1237725b2581ae47565adf287d"},"cell_type":"code","source":"plt.plot(hist.history['acc'])\nplt.plot(hist.history['val_acc'])\nplt.title('model accuracy')\nplt.ylabel('Accuracy')\nplt.xlabel('Epoch')\nplt.xlim(1,)\nplt.legend(['Train', 'Val'], loc='upper left')\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"4cf766dda339fe540538a1d4136b4ce0416d0026"},"cell_type":"code","source":"# summarize history for loss\nplt.plot(hist.history['loss'])\nplt.plot(hist.history['val_loss'])\nplt.title('Model Loss')\nplt.ylabel('Loss')\nplt.xlabel('Epoch')\nplt.xlim(1,)\nplt.legend(['Train', 'Val'], loc='upper left')\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d6cdda0e301be0b84e826e29f0f3a85c41aa6da9"},"cell_type":"markdown","source":"# Inference"},{"metadata":{"trusted":true,"_uuid":"b58cd95254f41e5002a17de0c3feab54a5fc3c67","scrolled":true},"cell_type":"code","source":"# prediction part\nbatch_size = 256\ndef batch_gen(test_df):\n    n_batches = math.ceil(len(test_df) / batch_size)\n    for i in range(n_batches):\n        texts = test_df.iloc[i*batch_size:(i+1)*batch_size, 1]\n        text_arr = np.array([text_to_array(text) for text in texts])\n        yield text_arr\n\n\n\nall_preds = []\nfor x in tqdm(batch_gen(test_df)):\n    all_preds.extend(model.predict(x).flatten())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"scrolled":true,"_uuid":"262740e511178ddff19e4eac3f33547b01a8dbb0"},"cell_type":"code","source":"%%time\nval_preds = []\nfor x in batch_gen(val_df):\n    val_preds.extend(model.predict(x).flatten())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c979185fbd9df86b85b9f8aea0aee99b5249177c"},"cell_type":"code","source":"gc.collect()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"fbaeaac1c2dcb4d9583b199c589d4e675e8abd70"},"cell_type":"code","source":"pd.Series(all_preds).describe()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"2940bf7a1e48456be9f3705619093ecf544817e7"},"cell_type":"code","source":"_thresh = [] \nfor thresh in np.arange(0.1, 0.501, 0.01): \n    _thresh.append([thresh, f1_score(val_df[\"target\"], (val_preds>thresh).astype(int))])\n    print(\"F1 score at threshold {0} is {1}\".format(thresh, f1_score(val_df[\"target\"], (val_preds>thresh).astype(int))))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"7ac7c5639ca95902743ed4515dcc3c79812b52ff"},"cell_type":"code","source":"_thresh = np.array(_thresh)\nbest_id = _thresh[:,1].argmax()\nbest_thresh = _thresh[best_id][0]\nbest_thresh","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"3e6ed54def110c881f401c6f8a844752baedcfbd"},"cell_type":"code","source":"y_te = (np.array(all_preds) > best_thresh).astype(np.int)\n\nsubmit_df = pd.DataFrame({\"qid\": test_df[\"qid\"], \"prediction\": y_te})\nsubmit_df.to_csv(\"submission.csv\", index=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d9ac3991dbd7cf22ab0d4fd981dace032ae38fd9"},"cell_type":"code","source":"submit_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"6e5a21b57d9f0944a4797abd9361f46a01ecab3f"},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"0c8bf17f9bccb7b0220380ac4a57959313695832"},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}