{"cells":[{"metadata":{"trusted":true,"_uuid":"3534aa98083481cfbefa98982f902c215753e493"},"cell_type":"code","source":"from keras.preprocessing.text import Tokenizer\nfrom keras.preprocessing.sequence import pad_sequences\n\nimport os\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nfrom tqdm import tqdm\nimport math\nfrom sklearn.model_selection import train_test_split","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"083d70c6848351af633b832751c75e5daa86ced1"},"cell_type":"code","source":"SEQ_LEN = 100  # magic number - length to truncate sequences of words","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"21f85dfd5b9bf3d8b27ba29149d52253e5d64049"},"cell_type":"markdown","source":"# Setup"},{"metadata":{"trusted":true,"_uuid":"78578eab64a477d0a5ad6b1c917ae154868a44df"},"cell_type":"code","source":"train_df = pd.read_csv(\"../input/train.csv\")\ntrain_df, val_df = train_test_split(train_df, test_size=0.07)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"05cab1c46136382e8b7ff2fa9d70fbb90063d8c2"},"cell_type":"code","source":"#minor eda: average question length (in words) is 12  , majority are under 12 words\ntrain_df.question_text.str.split().str.len().describe()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"92ffbf2ef35d2dc5863ee27c61eadd1869ca5440","scrolled":true},"cell_type":"code","source":"### Unclear why fails to open [encoding error], format is same as for glove. Will Debug, Dan:\n### f = open('../input/embeddings/paragram_300_sl999/paragram_300_sl999.txt')\n\n# embedding setup\n# Source https://blog.keras.io/using-pre-trained-word-embeddings-in-a-keras-model.html\n# \nembeddings_index = {}\nf = open('../input/embeddings/glove.840B.300d/glove.840B.300d.txt')\n# f = open('../input/embeddings/paragram_300_sl999/paragram_300_sl999.txt')\nfor line in tqdm(f):\n    values = line.split(\" \")\n    word = values[0]\n    coefs = np.asarray(values[1:], dtype='float32')\n    embeddings_index[word] = coefs\nf.close()\n\nprint('Found %s word vectors.' % len(embeddings_index))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f33f471ecaf004a58c8b950a81b1262fe646581c"},"cell_type":"code","source":"import re\n_WORD_SPLIT = re.compile(\"([.,!?\\\"':;)(])\")\n_DIGIT_RE = re.compile(br\"\\d\")\nSTOP_WORDS = \"\\\" \\' [ ] . , ! : ; ?\".split(\" \")\ndef basic_tokenizer(sentence):\n    \"\"\"Very basic tokenizer: split the sentence into a list of tokens.\"\"\"\n    words = []\n    for space_separated_fragment in sentence.strip().split():\n        words.extend(_WORD_SPLIT.split(space_separated_fragment))\n        # return [w.lower() for w in words if w not in stop_words and w != '' and w != ' ']\n    return [w.lower() for w in words if w != '' and w != ' ']","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"9cc8b0bcd285225ce651d024f506615d4656b9f7"},"cell_type":"code","source":"# Convert values to embeddings\ndef text_to_array(text):\n    empyt_emb = np.zeros(300)\n    text = basic_tokenizer(text[:-1])[:SEQ_LEN]\n    embeds = [embeddings_index.get(x, empyt_emb) for x in text]\n    embeds+= [empyt_emb] * (SEQ_LEN - len(embeds))\n    return np.array(embeds)\n\n# train_vects = [text_to_array(X_text) for X_text in tqdm(train_df[\"question_text\"])]\nval_vects = np.array([text_to_array(X_text) for X_text in tqdm(val_df[\"question_text\"][:3000])])\nval_y = np.array(val_df[\"target\"][:3000])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"0c950448e0717eaebb920d93cfdc6b4561e21853"},"cell_type":"code","source":"# Data providers\nbatch_size = 256\n\ndef batch_gen(train_df):\n    n_batches = math.ceil(len(train_df) / batch_size)\n    while True: \n        train_df = train_df.sample(frac=1.)  # Shuffle the data.\n        for i in range(n_batches):\n            texts = train_df.iloc[i*batch_size:(i+1)*batch_size, 1]\n            text_arr = np.array([text_to_array(text) for text in texts])\n            yield text_arr, np.array(train_df[\"target\"][i*batch_size:(i+1)*batch_size])","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"706c0224b6112e5a8f00ad25f35e90fbb9519a5f"},"cell_type":"markdown","source":"# Training"},{"metadata":{"trusted":true,"_uuid":"798c303ec834fb530a60a1e590cfbd9a86f93fde"},"cell_type":"code","source":"from keras.models import Sequential,Model\nfrom keras.layers import CuDNNLSTM, Dense, Bidirectional, Input,Dropout\n\nfrom keras import backend as K\nfrom keras.engine.topology import Layer\nfrom keras import initializers, regularizers, constraints","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"6f02eb4df531724fce73072dd97a637abbcff6e5","_kg_hide-input":false,"_kg_hide-output":false},"cell_type":"code","source":"# https://www.kaggle.com/qqgeogor/keras-lstm-attention-glove840b-lb-0-043\nclass Attention(Layer):\n    def __init__(self, step_dim,\n                 W_regularizer=None, b_regularizer=None,\n                 W_constraint=None, b_constraint=None,\n                 bias=True, **kwargs):\n        self.supports_masking = True\n        self.init = initializers.get('glorot_uniform')\n\n        self.W_regularizer = regularizers.get(W_regularizer)\n        self.b_regularizer = regularizers.get(b_regularizer)\n\n        self.W_constraint = constraints.get(W_constraint)\n        self.b_constraint = constraints.get(b_constraint)\n\n        self.bias = bias\n        self.step_dim = step_dim\n        self.features_dim = 0\n        super(Attention, self).__init__(**kwargs)\n\n    def build(self, input_shape):\n        assert len(input_shape) == 3\n\n        self.W = self.add_weight((input_shape[-1],),\n                                 initializer=self.init,\n                                 name='{}_W'.format(self.name),\n                                 regularizer=self.W_regularizer,\n                                 constraint=self.W_constraint)\n        self.features_dim = input_shape[-1]\n\n        if self.bias:\n            self.b = self.add_weight((input_shape[1],),\n                                     initializer='zero',\n                                     name='{}_b'.format(self.name),\n                                     regularizer=self.b_regularizer,\n                                     constraint=self.b_constraint)\n        else:\n            self.b = None\n\n        self.built = True\n\n    def compute_mask(self, input, input_mask=None):\n        return None\n\n    def call(self, x, mask=None):\n        features_dim = self.features_dim\n        step_dim = self.step_dim\n\n        eij = K.reshape(K.dot(K.reshape(x, (-1, features_dim)),\n                        K.reshape(self.W, (features_dim, 1))), (-1, step_dim))\n\n        if self.bias:\n            eij += self.b\n\n        eij = K.tanh(eij)\n\n        a = K.exp(eij)\n\n        if mask is not None:\n            a *= K.cast(mask, K.floatx())\n\n        a /= K.cast(K.sum(a, axis=1, keepdims=True) + K.epsilon(), K.floatx())\n\n        a = K.expand_dims(a)\n        weighted_input = x * a\n        return K.sum(weighted_input, axis=1)\n\n    def compute_output_shape(self, input_shape):\n        return input_shape[0],  self.features_dim","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"9ad418c95b31ab5691d49b72c7c9622ef9ea42cf"},"cell_type":"code","source":"inp = Input(shape=(SEQ_LEN,300 ))\nx = Bidirectional(CuDNNLSTM(64, return_sequences=True))(inp)\nx = Bidirectional(CuDNNLSTM(64,return_sequences=True))(x)\nx = Attention(SEQ_LEN)(x)\nx = Dense(256, activation=\"relu\")(x)\n# x = Dropout(0.25)(x)\nx = Dense(1, activation=\"sigmoid\")(x)\nmodel = Model(inputs=inp, outputs=x)\nmodel.compile(loss='binary_crossentropy',\n              optimizer='adam',\n              metrics=['accuracy'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"3f0b703ded691293450beb4ddf2d903d0e08c737"},"cell_type":"code","source":"mg = batch_gen(train_df)\nmodel.fit_generator(mg, epochs=20,\n                    steps_per_epoch=1000,\n                    validation_data=(val_vects, val_y),\n                    verbose=True)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d6cdda0e301be0b84e826e29f0f3a85c41aa6da9"},"cell_type":"markdown","source":"# Inference"},{"metadata":{"trusted":true,"_uuid":"b58cd95254f41e5002a17de0c3feab54a5fc3c67"},"cell_type":"code","source":"# prediction part\nbatch_size = 256\ndef batch_gen(test_df):\n    n_batches = math.ceil(len(test_df) / batch_size)\n    for i in range(n_batches):\n        texts = test_df.iloc[i*batch_size:(i+1)*batch_size, 1]\n        text_arr = np.array([text_to_array(text) for text in texts])\n        yield text_arr\n\ntest_df = pd.read_csv(\"../input/test.csv\")\n\nall_preds = []\nfor x in tqdm(batch_gen(test_df)):\n    all_preds.extend(model.predict(x).flatten())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"3e6ed54def110c881f401c6f8a844752baedcfbd"},"cell_type":"code","source":"y_te = (np.array(all_preds) > 0.35).astype(np.int)\n\nsubmit_df = pd.DataFrame({\"qid\": test_df[\"qid\"], \"prediction\": y_te})\nsubmit_df.to_csv(\"submission.csv\", index=False)","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}