{"cells":[{"metadata":{"_uuid":"8c102712b233a094a1fa3a771e0b518b2d03dd7f"},"cell_type":"markdown","source":"# refference\n[mix-of-nn-models-based-on-meta-embedding](https://www.kaggle.com/shujian/mix-of-nn-models-based-on-meta-embedding)  \n[a-look-at-different-embeddings](https://www.kaggle.com/sudalairajkumar/a-look-at-different-embeddings)  \n[cnn-in-keras-with-pretrained-word2vec-weights](https://www.kaggle.com/marijakekic/cnn-in-keras-with-pretrained-word2vec-weights)  \n[how-to-preprocessing-when-using-embeddings](https://www.kaggle.com/christofhenkel/how-to-preprocessing-when-using-embeddings)  "},{"metadata":{"trusted":true,"_uuid":"5f7789df06861b5e35ee9c332a894892df12ca82"},"cell_type":"code","source":"import os\nimport time\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nfrom tqdm import tqdm\nimport math\nfrom sklearn.model_selection import train_test_split\nfrom sklearn import metrics\nfrom gensim.models import KeyedVectors\n\nfrom keras.preprocessing.text import Tokenizer\nfrom keras.preprocessing.sequence import pad_sequences\nfrom keras.layers import Dense, Input, CuDNNLSTM, Embedding, Dropout, Activation, CuDNNGRU, Conv1D\nfrom keras.layers import Bidirectional, GlobalMaxPool1D, GlobalMaxPooling1D, GlobalAveragePooling1D\nfrom keras.layers import Input, Embedding, Dense, Conv2D, MaxPool2D, concatenate\nfrom keras.layers import Reshape, Flatten, Concatenate, Dropout, SpatialDropout1D\nfrom keras.optimizers import Adam\nfrom keras.models import Model\nfrom keras import backend as K\nfrom keras.engine.topology import Layer\nfrom keras import initializers, regularizers, constraints, optimizers, layers","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nfrom tqdm import tqdm\ntqdm.pandas()\nimport os\nprint(os.listdir(\"../input\"))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"39aa0c7f6b1387bc254a68613772c4034bddc10c"},"cell_type":"code","source":"def clean_text(x):\n    x = str(x)\n    for punct in \"/-'\":\n        x = x.replace(punct, ' ')\n    for punct in '&':\n        x = x.replace(punct, f' {punct} ')\n    for punct in '?!.,\"#$%\\'()*+-/:;<=>@[\\\\]^_`{|}~' + '“”’':\n        x = x.replace(punct, '')\n    return x","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"cfe0a60ec23b891dd1b0c1037d1bc508837f61a3"},"cell_type":"code","source":"import re\ndef clean_numbers(x):\n    x = re.sub('[0-9]{5,}', '#####', x)\n    x = re.sub('[0-9]{4}', '####', x)\n    x = re.sub('[0-9]{3}', '###', x)\n    x = re.sub('[0-9]{2}', '##', x)\n    return x","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"299e248362b0e4705465cfc73fb578395db72c0a"},"cell_type":"code","source":"def _get_mispell(mispell_dict):\n    mispell_re = re.compile('(%s)' % '|'.join(mispell_dict.keys()))\n    return mispell_dict, mispell_re\n\nmispell_dict = {'colour':'color',\n                'centre':'center',\n                'didnt':'did not',\n                'doesnt':'does not',\n                'isnt':'is not',\n                'shouldnt':'should not',\n                'favourite':'favorite',\n                'travelling':'traveling',\n                'counselling':'counseling',\n                'theatre':'theater',\n                'cancelled':'canceled',\n                'labour':'labor',\n                'organisation':'organization',\n                'wwii':'world war 2',\n                'citicise':'criticize',\n                'instagram': 'social medium',\n                'whatsapp': 'social medium',\n                'snapchat': 'social medium'\n\n                }\nmispellings, mispellings_re = _get_mispell(mispell_dict)\n\ndef replace_typical_misspell(text):\n    def replace(match):\n        return mispellings[match.group(0)]\n\n    return mispellings_re.sub(replace, text)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"3c5a88e2efbcd4e512660b098a82b9475c8d4c75"},"cell_type":"code","source":"def cleansing(text):\n    text[\"question_text\"] = text[\"question_text\"].progress_apply(lambda x: clean_text(x))\n    text[\"question_text\"] = text[\"question_text\"].progress_apply(lambda x: clean_numbers(x))\n    text[\"question_text\"] = text[\"question_text\"].progress_apply(lambda x: replace_typical_misspell(x))\n    sentences = text[\"question_text\"].progress_apply(lambda x: x.split())\n    to_remove = ['a','to','of','and']\n    sentences = [[word for word in sentence if not word in to_remove] for sentence in tqdm(sentences)]\n    \n    return sentences","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e31d6e126881ee56a1de3efe02fcf309e900ef00"},"cell_type":"code","source":"## some config values \nembed_size = 300 # how big is each word vector\nmax_features = 95000 # how many unique words to use (i.e num rows in embedding vector)\nmaxlen = 70 # max number of words in a question to use","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"5cdc95950037613c690c49b27930ae0f59eb23c3"},"cell_type":"code","source":"def load_and_prec():\n    train_df = pd.read_csv(\"../input/train.csv\")\n    test_df = pd.read_csv(\"../input/test.csv\")\n    print(\"Train shape : \",train_df.shape)\n    print(\"Test shape : \",test_df.shape)\n    \n    ## split to train and val\n    train_df, val_df = train_test_split(train_df, test_size=0.08, random_state=2018)\n\n    ## fill up the missing values\n    train_X = cleansing(train_df)\n    val_X = cleansing(val_df)\n    test_X = cleansing(test_df)\n\n    ## Tokenize the sentences\n    tokenizer = Tokenizer(num_words=max_features)\n    tokenizer.fit_on_texts(list(train_X))\n    train_X = tokenizer.texts_to_sequences(train_X)\n    val_X = tokenizer.texts_to_sequences(val_X)\n    test_X = tokenizer.texts_to_sequences(test_X)\n\n    ## Pad the sentences \n    train_X = pad_sequences(train_X, maxlen=maxlen)\n    val_X = pad_sequences(val_X, maxlen=maxlen)\n    test_X = pad_sequences(test_X, maxlen=maxlen)\n\n    ## Get the target values\n    train_y = train_df['target'].values\n    val_y = val_df['target'].values  \n    \n    #shuffling the data\n    np.random.seed(2018)\n    trn_idx = np.random.permutation(len(train_X))\n    val_idx = np.random.permutation(len(val_X))\n\n    train_X = train_X[trn_idx]\n    val_X = val_X[val_idx]\n    train_y = train_y[trn_idx]\n    val_y = val_y[val_idx]    \n    \n    return train_X, val_X, test_X, train_y, val_y, tokenizer.word_index","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a662716cc5fbbcc0c84019a87c52332ed8912e8d"},"cell_type":"code","source":"def load_glove(word_index):\n    EMBEDDING_FILE = '../input/embeddings/glove.840B.300d/glove.840B.300d.txt'\n    def get_coefs(word,*arr): return word, np.asarray(arr, dtype='float32')\n    embeddings_index = dict(get_coefs(*o.split(\" \")) for o in open(EMBEDDING_FILE))\n\n    all_embs = np.stack(embeddings_index.values())\n    emb_mean,emb_std = all_embs.mean(), all_embs.std()\n    embed_size = all_embs.shape[1]\n\n    # word_index = tokenizer.word_index\n    nb_words = min(max_features, len(word_index))\n    embedding_matrix = np.random.normal(emb_mean, emb_std, (nb_words, embed_size))\n    for word, i in word_index.items():\n        if i >= max_features: continue\n        embedding_vector = embeddings_index.get(word)\n        if embedding_vector is not None: embedding_matrix[i] = embedding_vector\n            \n    return embedding_matrix \n    \ndef load_fasttext(word_index):    \n    EMBEDDING_FILE = '../input/embeddings/wiki-news-300d-1M/wiki-news-300d-1M.vec'\n    def get_coefs(word,*arr): return word, np.asarray(arr, dtype='float32')\n    embeddings_index = dict(get_coefs(*o.split(\" \")) for o in open(EMBEDDING_FILE) if len(o)>100)\n\n    all_embs = np.stack(embeddings_index.values())\n    emb_mean,emb_std = all_embs.mean(), all_embs.std()\n    embed_size = all_embs.shape[1]\n\n    # word_index = tokenizer.word_index\n    nb_words = min(max_features, len(word_index))\n    embedding_matrix = np.random.normal(emb_mean, emb_std, (nb_words, embed_size))\n    for word, i in word_index.items():\n        if i >= max_features: continue\n        embedding_vector = embeddings_index.get(word)\n        if embedding_vector is not None: embedding_matrix[i] = embedding_vector\n\n    return embedding_matrix\n\ndef load_para(word_index):\n    EMBEDDING_FILE = '../input/embeddings/paragram_300_sl999/paragram_300_sl999.txt'\n    def get_coefs(word,*arr): return word, np.asarray(arr, dtype='float32')\n    embeddings_index = dict(get_coefs(*o.split(\" \")) for o in open(EMBEDDING_FILE, encoding=\"utf8\", errors='ignore') if len(o)>100)\n\n    all_embs = np.stack(embeddings_index.values())\n    emb_mean,emb_std = all_embs.mean(), all_embs.std()\n    embed_size = all_embs.shape[1]\n\n    # word_index = tokenizer.word_index\n    nb_words = min(max_features, len(word_index))\n    embedding_matrix = np.random.normal(emb_mean, emb_std, (nb_words, embed_size))\n    for word, i in word_index.items():\n        if i >= max_features: continue\n        embedding_vector = embeddings_index.get(word)\n        if embedding_vector is not None: embedding_matrix[i] = embedding_vector\n    \n    return embedding_matrix\n\ndef load_google(word_index):\n    news_path = '../input/embeddings/GoogleNews-vectors-negative300/GoogleNews-vectors-negative300.bin'\n    embeddings_index = KeyedVectors.load_word2vec_format(news_path, binary=True)\n    vocabulary_size=min(max_features, len(word_index)+1)\n    embedding_matrix = np.zeros((vocabulary_size, embed_size))\n    for word, i in word_index.items():\n        if i>=max_features:\n            continue\n        try:\n            embedding_vector = embeddings_index[word]\n            embedding_matrix[i] = embedding_vector\n        except KeyError:\n            embedding_matrix[i]=np.random.normal(0,np.sqrt(0.25),embed_size)\n            \n    return embedding_matrix","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"1f72c8c9573fb840ceb50a9cd4ac4e455e1c0ea7"},"cell_type":"code","source":"# https://www.kaggle.com/yekenot/2dcnn-textclassifier\ndef model_cnn(embedding_matrix):\n    filter_sizes = [1,2,3,5]\n    num_filters = 36\n\n    inp = Input(shape=(maxlen,))\n    x = Embedding(max_features, embed_size, weights=[embedding_matrix])(inp)\n    x = Reshape((maxlen, embed_size, 1))(x)\n\n    maxpool_pool = []\n    for i in range(len(filter_sizes)):\n        conv = Conv2D(num_filters, kernel_size=(filter_sizes[i], embed_size),\n                                     kernel_initializer='he_normal', activation='relu')(x)\n        maxpool_pool.append(MaxPool2D(pool_size=(maxlen - filter_sizes[i] + 1, 1))(conv))\n\n    z = Concatenate(axis=1)(maxpool_pool)   \n    z = Flatten()(z)\n    z = Dropout(0.1)(z)\n\n    outp = Dense(1, activation=\"sigmoid\")(z)\n\n    model = Model(inputs=inp, outputs=outp)\n    model.compile(loss='binary_crossentropy', optimizer='adam', metrics=['accuracy'])\n    \n    return model","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"84e00df2c7b94205f5588af503f62412c48f46f3"},"cell_type":"code","source":"# https://www.kaggle.com/suicaokhoailang/lstm-attention-baseline-0-652-lb\n\nclass Attention(Layer):\n    def __init__(self, step_dim,\n                 W_regularizer=None, b_regularizer=None,\n                 W_constraint=None, b_constraint=None,\n                 bias=True, **kwargs):\n        self.supports_masking = True\n        self.init = initializers.get('glorot_uniform')\n\n        self.W_regularizer = regularizers.get(W_regularizer)\n        self.b_regularizer = regularizers.get(b_regularizer)\n\n        self.W_constraint = constraints.get(W_constraint)\n        self.b_constraint = constraints.get(b_constraint)\n\n        self.bias = bias\n        self.step_dim = step_dim\n        self.features_dim = 0\n        super(Attention, self).__init__(**kwargs)\n\n    def build(self, input_shape):\n        assert len(input_shape) == 3\n\n        self.W = self.add_weight((input_shape[-1],),\n                                 initializer=self.init,\n                                 name='{}_W'.format(self.name),\n                                 regularizer=self.W_regularizer,\n                                 constraint=self.W_constraint)\n        self.features_dim = input_shape[-1]\n\n        if self.bias:\n            self.b = self.add_weight((input_shape[1],),\n                                     initializer='zero',\n                                     name='{}_b'.format(self.name),\n                                     regularizer=self.b_regularizer,\n                                     constraint=self.b_constraint)\n        else:\n            self.b = None\n\n        self.built = True\n\n    def compute_mask(self, input, input_mask=None):\n        return None\n\n    def call(self, x, mask=None):\n        features_dim = self.features_dim\n        step_dim = self.step_dim\n\n        eij = K.reshape(K.dot(K.reshape(x, (-1, features_dim)),\n                        K.reshape(self.W, (features_dim, 1))), (-1, step_dim))\n\n        if self.bias:\n            eij += self.b\n\n        eij = K.tanh(eij)\n\n        a = K.exp(eij)\n\n        if mask is not None:\n            a *= K.cast(mask, K.floatx())\n\n        a /= K.cast(K.sum(a, axis=1, keepdims=True) + K.epsilon(), K.floatx())\n\n        a = K.expand_dims(a)\n        weighted_input = x * a\n        return K.sum(weighted_input, axis=1)\n\n    def compute_output_shape(self, input_shape):\n        return input_shape[0],  self.features_dim","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"05164d541a0c35cae727d0338548d156efe21427"},"cell_type":"code","source":"def model_lstm_atten(embedding_matrix):\n    inp = Input(shape=(maxlen,))\n    x = Embedding(max_features, embed_size, weights=[embedding_matrix], trainable=False)(inp)\n    x = Bidirectional(CuDNNLSTM(128, return_sequences=True))(x)\n    x = Bidirectional(CuDNNLSTM(64, return_sequences=True))(x)\n    x = Attention(maxlen)(x)\n    x = Dense(64, activation=\"relu\")(x)\n    x = Dense(1, activation=\"sigmoid\")(x)\n    model = Model(inputs=inp, outputs=x)\n    model.compile(loss='binary_crossentropy', optimizer='adam', metrics=['accuracy'])\n    \n    return model","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"2bf953e2b5b6d9363eda89971ecc0ac416e2ddd0"},"cell_type":"code","source":"def model_gru_srk_atten(embedding_matrix):\n    inp = Input(shape=(maxlen,))\n    x = Embedding(max_features, embed_size, weights=[embedding_matrix])(inp)\n    x = Bidirectional(CuDNNGRU(64, return_sequences=True))(x)\n    x = Attention(maxlen)(x) # New\n    x = Dense(16, activation=\"relu\")(x)\n    x = Dropout(0.1)(x)\n    x = Dense(1, activation=\"sigmoid\")(x)\n    model = Model(inputs=inp, outputs=x)\n    model.compile(loss='binary_crossentropy', optimizer='adam', metrics=['accuracy'])\n    \n    return model    \n    ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"17b13ef39fbbf1919307c23efd516eddc2135023"},"cell_type":"code","source":"def model_lstm_du(embedding_matrix):\n    inp = Input(shape=(maxlen,))\n    x = Embedding(max_features, embed_size, weights=[embedding_matrix])(inp)\n    x = Bidirectional(CuDNNGRU(64, return_sequences=True))(x)\n    avg_pool = GlobalAveragePooling1D()(x)\n    max_pool = GlobalMaxPooling1D()(x)\n    conc = concatenate([avg_pool, max_pool])\n    conc = Dense(64, activation=\"relu\")(conc)\n    conc = Dropout(0.1)(conc)\n    outp = Dense(1, activation=\"sigmoid\")(conc)\n    \n    model = Model(inputs=inp, outputs=outp)\n    model.compile(loss='binary_crossentropy', optimizer='adam', metrics=['accuracy'])\n    return model","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c02ace553c962a10f038d21801c8b46f567a0c3f"},"cell_type":"code","source":"def model_gru_atten_3(embedding_matrix):\n    inp = Input(shape=(maxlen,))\n    x = Embedding(max_features, embed_size, weights=[embedding_matrix], trainable=False)(inp)\n    x = Bidirectional(CuDNNGRU(128, return_sequences=True))(x)\n    x = Bidirectional(CuDNNGRU(100, return_sequences=True))(x)\n    x = Bidirectional(CuDNNGRU(64, return_sequences=True))(x)\n    x = Attention(maxlen)(x)\n    x = Dense(1, activation=\"sigmoid\")(x)\n    model = Model(inputs=inp, outputs=x)\n    model.compile(loss='binary_crossentropy', optimizer='adam', metrics=['accuracy'])\n    \n    return model","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e8523d876b6eae762e673b777cc7af4d7f085792"},"cell_type":"code","source":"# https://www.kaggle.com/strideradu/word2vec-and-gensim-go-go-go\ndef train_pred(model, epochs=2):\n    for e in range(epochs):\n        model.fit(train_X, train_y, batch_size=512, epochs=1, validation_data=(val_X, val_y))\n        pred_val_y = model.predict([val_X], batch_size=1024, verbose=0)\n\n        best_thresh = 0.5\n        best_score = 0.0\n        for thresh in np.arange(0.1, 0.501, 0.01):\n            thresh = np.round(thresh, 2)\n            score = metrics.f1_score(val_y, (pred_val_y > thresh).astype(int))\n            if score > best_score:\n                best_thresh = thresh\n                best_score = score\n\n        print(\"Val F1 Score: {:.4f}\".format(best_score))\n\n    pred_test_y = model.predict([test_X], batch_size=1024, verbose=0)\n    return pred_val_y, pred_test_y, best_score","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"99d03d2eb63600f1b222522616eab3fa35819f37"},"cell_type":"code","source":"train_X, val_X, test_X, train_y, val_y, word_index = load_and_prec()\nembedding_matrix_1 = load_glove(word_index)\nembedding_matrix_2 = load_fasttext(word_index)\nembedding_matrix_3 = load_para(word_index)\nembedding_matrix_4 = load_google(word_index)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f0ea9b1468bd7cd3ceead2593c641900dc3a2a77"},"cell_type":"code","source":"embedding_matrix = np.mean([embedding_matrix_1, embedding_matrix_2, embedding_matrix_3, embedding_matrix_4], axis = 0)\n#embedding_matrix = np.mean([embedding_matrix_1, embedding_matrix_3], axis = 0)\nnp.shape(embedding_matrix)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"da641bf6a7e0a40863e571c78eba9285fd8671b0"},"cell_type":"code","source":"outputs = []","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"8b0ce7a68d32a893affb0f3e57e0d1517776e1b1"},"cell_type":"code","source":"pred_val_y, pred_test_y, best_score = train_pred(model_gru_atten_3(embedding_matrix), epochs = 2)\noutputs.append([pred_val_y, pred_test_y, best_score, '3 GRU w/ atten'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"7aae47c7ded0b4f4849fe68b8c2f282ea03e1d20"},"cell_type":"code","source":"pred_val_y, pred_test_y, best_score = train_pred(model_gru_srk_atten(embedding_matrix), epochs = 2)\noutputs.append([pred_val_y, pred_test_y, best_score, 'gru atten srk'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"40ec64ad93aea2eb4d303fb7c1e9fc58590885dc"},"cell_type":"code","source":"pred_val_y, pred_test_y, best_score = train_pred(model_cnn(embedding_matrix), epochs = 3)\noutputs.append([pred_val_y, pred_test_y, best_score, '2d CNN'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"2d5802cdc286219cbddd4caa9851d08cdf39ebf6"},"cell_type":"code","source":"pred_val_y, pred_test_y, best_score = train_pred(model_cnn(embedding_matrix_1), epochs = 2) # GloVe only\noutputs.append([pred_val_y, pred_test_y, best_score, '2d CNN GloVe'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"2ce89a82185e091178a3878fb87deddba8e7a381"},"cell_type":"code","source":"pred_val_y, pred_test_y, best_score = train_pred(model_lstm_du(embedding_matrix), epochs = 2)\noutputs.append([pred_val_y, pred_test_y, best_score, 'LSTM DU'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"57af85779fca0bfebc26813de4f4d07137e68510"},"cell_type":"code","source":"pred_val_y, pred_test_y, best_score = train_pred(model_lstm_atten(embedding_matrix), epochs = 2)\noutputs.append([pred_val_y, pred_test_y, best_score, '2 LSTM w/ attention'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d7865acc1dfd4074ad9736f0563c7a53eab2bdfc"},"cell_type":"code","source":"pred_val_y, pred_test_y, best_score = train_pred(model_lstm_atten(embedding_matrix_1), epochs = 2) # Only GloVe\noutputs.append([pred_val_y, pred_test_y, best_score, '2 LSTM w/ attention GloVe'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"0ced4a3ed8315ff92cad4ea256f9ac6703909833"},"cell_type":"code","source":"pred_val_y, pred_test_y, best_score = train_pred(model_lstm_atten(embedding_matrix_2), epochs = 2) # Only Wiki\noutputs.append([pred_val_y, pred_test_y, best_score, '2 LSTM w/ attention Wiki'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"6d0d468740e54fa4ae52262a54179e58b64ef7da"},"cell_type":"code","source":"pred_val_y, pred_test_y, best_score = train_pred(model_lstm_atten(embedding_matrix_3), epochs = 2) # Only Para\noutputs.append([pred_val_y, pred_test_y, best_score, '2 LSTM w/ attention Para'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"af32127806a14a794f256f0862312b7b29b3753c"},"cell_type":"code","source":"pred_val_y, pred_test_y, best_score = train_pred(model_lstm_atten(embedding_matrix_4), epochs = 2) # Only Google\noutputs.append([pred_val_y, pred_test_y, best_score, '2 LSTM w/ attention Google'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"74f090f4c169d27c6c198f8faa46ffa077029953"},"cell_type":"code","source":"outputs.sort(key=lambda x: x[2]) # Sort the output by val f1 score\nweights = [i for i in range(1, len(outputs) + 1)]\nweights = [float(i) / sum(weights) for i in weights] \nprint(weights)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"144b69d8403ef889024bf7d7bb3ace8fd5b20b9e"},"cell_type":"code","source":"for output in outputs:\n    print(output[2], output[3])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"988f260e9aaecbadb82ac15abc692971589de0fa"},"cell_type":"code","source":"# pred_val_y = np.sum([outputs[i][0] * weights[i] for i in range(len(outputs))], axis = 0)\npred_val_y = np.mean([outputs[i][0] for i in range(len(outputs))], axis = 0) # to avoid overfitting, just take average\n\nthresholds = []\nfor thresh in np.arange(0.1, 0.501, 0.01):\n    thresh = np.round(thresh, 2)\n    res = metrics.f1_score(val_y, (pred_val_y > thresh).astype(int))\n    thresholds.append([thresh, res])\n    print(\"F1 score at threshold {0} is {1}\".format(thresh, res))\n    \nthresholds.sort(key=lambda x: x[1], reverse=True)\nbest_thresh = thresholds[0][0]\nprint(\"Best threshold: \", best_thresh)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"74a84952d3a6a2b265dcd72a1ba8f41e315df332","scrolled":true},"cell_type":"code","source":"# pred_test_y = np.sum([outputs[i][1] * weights[i] for i in range(len(outputs))], axis = 0)\npred_test_y = np.mean([outputs[i][1] for i in range(len(outputs))], axis = 0)\n\npred_test_y = (pred_test_y > best_thresh).astype(int)\ntest_df = pd.read_csv(\"../input/test.csv\", usecols=[\"qid\"])\nout_df = pd.DataFrame({\"qid\":test_df[\"qid\"].values})\nout_df['prediction'] = pred_test_y\nout_df.to_csv(\"submission.csv\", index=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"af44b57ac7afcc8552dd0de65dcecb26538cfc2b"},"cell_type":"markdown","source":""}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}