{"cells":[{"metadata":{"_uuid":"d65f57d963f9cbb340d5d3dcfa70b495734efc0a"},"cell_type":"markdown","source":"I was trying to clean some of my code so I can add more models. However, this can never happen without the awesome kernels from other talented Kagglers. Forgive me if I missed any.\n\n* Snapshot ensemble from https://github.com/arthurdouillard/keras-snapshot_ensembles\n* Chenglong Chen used snapshot ensemble in Mercari: https://github.com/ChenglongChen/tensorflow-XNN\n\n* https://www.kaggle.com/hireme/fun-api-keras-f1-metric-cyclical-learning-rate/code\n* Based on SRK's kernel: https://www.kaggle.com/sudalairajkumar/a-look-at-different-embeddings\n* Vladimir Demidov's 2DCNN textClassifier: https://www.kaggle.com/yekenot/2dcnn-textclassifier\n* Attention layer from Khoi Ngyuen: https://www.kaggle.com/suicaokhoailang/lstm-attention-baseline-0-652-lb\n* LSTM model from Strideradu: https://www.kaggle.com/strideradu/word2vec-and-gensim-go-go-go\n* https://www.kaggle.com/danofer/different-embeddings-with-attention-fork\n* https://www.kaggle.com/ryanzhang/tfidf-naivebayes-logreg-baseline\n* Borrowed some idea from this model: https://www.kaggle.com/c/jigsaw-toxic-comment-classification-challenge/discussion/52644\n* Sentence length seems to a good feature: https://www.kaggle.com/thebrownviking20/analyzing-quora-for-the-insinceres\n\nSome new things here:\n\n* Take average of embeddings (Unweighted DME) instead of blending predictions: https://arxiv.org/pdf/1804.07983.pdf\n* The original paper of this idea comes from: Frustratingly Easy Meta-Embedding – Computing Meta-Embeddings by Averaging Source Word Embeddings\n* Modified the code to choose best threshold\n* Robust method for blending weights: sort the val score and give the final weight\n\nSome thoughts:\n\n* Although I pulished a kernel on Transformer, I will not use it\n* Too much randomness in CuDNN. You may get different results by just rerunning this kernel\n* Blending rocks"},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nimport os\nprint(os.listdir(\"../input\"))\n\n# Any results you write to the current directory are saved as output.","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e31d6e126881ee56a1de3efe02fcf309e900ef00"},"cell_type":"code","source":"## some config values \nembed_size = 300 # how big is each word vector\nmax_features = 95000 # how many unique words to use (i.e num rows in embedding vector)\nmaxlen = 70 # max number of words in a question to use","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"522d9790478f62193ea5c315372a2ab9cbe9b27f"},"cell_type":"markdown","source":"**Load packages and data**"},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"import os\nimport time\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nfrom tqdm import tqdm\nimport math\nfrom sklearn.model_selection import train_test_split\nfrom sklearn import metrics\nfrom sklearn.model_selection import GridSearchCV, StratifiedKFold\nfrom sklearn.metrics import f1_score, roc_auc_score\n\nfrom keras.preprocessing.text import Tokenizer\nfrom keras.preprocessing.sequence import pad_sequences\nfrom keras.layers import Dense, Input, CuDNNLSTM, Embedding, Dropout, Activation, CuDNNGRU, Conv1D\nfrom keras.layers import Bidirectional, GlobalMaxPool1D, GlobalMaxPooling1D, GlobalAveragePooling1D\nfrom keras.layers import Input, Embedding, Dense, Conv2D, MaxPool2D, concatenate\nfrom keras.layers import Reshape, Flatten, Concatenate, Dropout, SpatialDropout1D\nfrom keras.optimizers import Adam\nfrom keras.models import Model\nfrom keras import backend as K\nfrom keras.engine.topology import Layer\nfrom keras import initializers, regularizers, constraints, optimizers, layers\nfrom keras.layers import concatenate\nfrom keras.optimizers import SGD","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"5cdc95950037613c690c49b27930ae0f59eb23c3"},"cell_type":"code","source":"def load_and_prec():\n    train_df = pd.read_csv(\"../input/train.csv\")\n    test_df = pd.read_csv(\"../input/test.csv\")\n    print(\"Train shape : \",train_df.shape)\n    print(\"Test shape : \",test_df.shape)\n    \n    ## fill up the missing values\n    train_X = train_df[\"question_text\"].fillna(\"_##_\").values\n    test_X = test_df[\"question_text\"].fillna(\"_##_\").values\n\n    ## Tokenize the sentences\n    tokenizer = Tokenizer(num_words=max_features)\n    tokenizer.fit_on_texts(list(train_X))\n    train_X = tokenizer.texts_to_sequences(train_X)\n    test_X = tokenizer.texts_to_sequences(test_X)\n\n    ## Pad the sentences \n    train_X = pad_sequences(train_X, maxlen=maxlen)\n    test_X = pad_sequences(test_X, maxlen=maxlen)\n\n    ## Get the target values\n    train_y = train_df['target'].values\n    \n    #shuffling the data\n    np.random.seed(2018)\n    trn_idx = np.random.permutation(len(train_X))\n\n    train_X = train_X[trn_idx]\n    train_y = train_y[trn_idx]\n    \n    return train_X, test_X, train_y, tokenizer.word_index","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"dba1893c267a1e7536bbf720636647d85c7e349c"},"cell_type":"markdown","source":"**Load embeddings**"},{"metadata":{"trusted":true,"_uuid":"a662716cc5fbbcc0c84019a87c52332ed8912e8d"},"cell_type":"code","source":"def load_glove(word_index):\n    EMBEDDING_FILE = '../input/embeddings/glove.840B.300d/glove.840B.300d.txt'\n    def get_coefs(word,*arr): return word, np.asarray(arr, dtype='float32')\n    embeddings_index = dict(get_coefs(*o.split(\" \")) for o in open(EMBEDDING_FILE))\n\n    all_embs = np.stack(embeddings_index.values())\n    emb_mean,emb_std = all_embs.mean(), all_embs.std()\n    embed_size = all_embs.shape[1]\n\n    # word_index = tokenizer.word_index\n    nb_words = min(max_features, len(word_index))\n    embedding_matrix = np.random.normal(emb_mean, emb_std, (nb_words, embed_size))\n    for word, i in word_index.items():\n        if i >= max_features: continue\n        embedding_vector = embeddings_index.get(word)\n        if embedding_vector is not None: embedding_matrix[i] = embedding_vector\n            \n    return embedding_matrix \n    \ndef load_fasttext(word_index):    \n    EMBEDDING_FILE = '../input/embeddings/wiki-news-300d-1M/wiki-news-300d-1M.vec'\n    def get_coefs(word,*arr): return word, np.asarray(arr, dtype='float32')\n    embeddings_index = dict(get_coefs(*o.split(\" \")) for o in open(EMBEDDING_FILE) if len(o)>100)\n\n    all_embs = np.stack(embeddings_index.values())\n    emb_mean,emb_std = all_embs.mean(), all_embs.std()\n    embed_size = all_embs.shape[1]\n\n    # word_index = tokenizer.word_index\n    nb_words = min(max_features, len(word_index))\n    embedding_matrix = np.random.normal(emb_mean, emb_std, (nb_words, embed_size))\n    for word, i in word_index.items():\n        if i >= max_features: continue\n        embedding_vector = embeddings_index.get(word)\n        if embedding_vector is not None: embedding_matrix[i] = embedding_vector\n\n    return embedding_matrix\n\ndef load_para(word_index):\n    EMBEDDING_FILE = '../input/embeddings/paragram_300_sl999/paragram_300_sl999.txt'\n    def get_coefs(word,*arr): return word, np.asarray(arr, dtype='float32')\n    embeddings_index = dict(get_coefs(*o.split(\" \")) for o in open(EMBEDDING_FILE, encoding=\"utf8\", errors='ignore') if len(o)>100)\n\n    all_embs = np.stack(embeddings_index.values())\n    emb_mean,emb_std = all_embs.mean(), all_embs.std()\n    embed_size = all_embs.shape[1]\n\n    # word_index = tokenizer.word_index\n    nb_words = min(max_features, len(word_index))\n    embedding_matrix = np.random.normal(emb_mean, emb_std, (nb_words, embed_size))\n    for word, i in word_index.items():\n        if i >= max_features: continue\n        embedding_vector = embeddings_index.get(word)\n        if embedding_vector is not None: embedding_matrix[i] = embedding_vector\n    \n    return embedding_matrix","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"5333fb187ba56ea9a30cceb6d70fc8a71c02bf4a"},"cell_type":"markdown","source":"**Snapshot ensemble**"},{"metadata":{"trusted":true,"_uuid":"8face0155bcd68d52b8dc4228d91049343dc6dd3"},"cell_type":"code","source":"# https://github.com/arthurdouillard/keras-snapshot_ensembles/blob/master/snapshot.py\nimport math\nimport os\nimport glob\n\nfrom keras import backend as K\nfrom keras.layers import *\nfrom keras.models import Model\nfrom keras.callbacks import Callback\n\n\nclass Snapshot(Callback):\n\n    def __init__(self, folder_path, nb_epochs, nb_cycles=5, verbose=0):\n        if nb_cycles > nb_epochs:\n            raise ValueError('nb_epochs has to be lower than nb_cycles.')\n\n        super(Snapshot, self).__init__()\n        self.verbose = verbose\n        self.folder_path = folder_path\n        self.nb_epochs = nb_epochs\n        self.nb_cycles = nb_cycles\n        self.period = self.nb_epochs // self.nb_cycles\n        self.nb_digits = len(str(self.nb_cycles))\n        self.path_format = os.path.join(self.folder_path, 'weights_cycle_{}.h5')\n\n\n    def on_epoch_end(self, epoch, logs=None):\n        if epoch == 0 or (epoch + 1) % self.period != 0: return\n        # Only save at the end of a cycle, a not at the beginning\n\n        if not os.path.exists(self.folder_path):\n            os.makedirs(self.folder_path)\n\n        cycle = int(epoch / self.period)\n        cycle_str = str(cycle).rjust(self.nb_digits, '0')\n        self.model.save_weights(self.path_format.format(cycle_str), overwrite=True)\n\n        # Resetting the learning rate\n        K.set_value(self.model.optimizer.lr, self.base_lr)\n\n        if self.verbose > 0:\n            print('\\nEpoch %05d: Reached %d-th cycle, saving model.' % (epoch, cycle))\n\n\n    def on_epoch_begin(self, epoch, logs=None):\n        if epoch <= 0: return\n\n        lr = self.schedule(epoch)\n        K.set_value(self.model.optimizer.lr, lr)\n\n        if self.verbose > 0:\n            print('\\nEpoch %05d: Snapchot modifying learning '\n                  'rate to %s.' % (epoch + 1, lr))\n\n\n    def set_model(self, model):\n        self.model = model\n        if not hasattr(self.model.optimizer, 'lr'):\n            raise ValueError('Optimizer must have a \"lr\" attribute.')\n\n        # Get initial learning rate\n        self.base_lr = float(K.get_value(self.model.optimizer.lr))\n\n\n    def schedule(self, epoch):\n        lr = math.pi * (epoch % self.period) / self.period\n        lr = self.base_lr / 2 * (math.cos(lr) + 1)\n        return lr\n    \n\n\ndef load_ensemble(folder, keep_last=None):\n    paths = glob.glob(os.path.join(folder, 'weights_cycle_*.h5'))\n    print('Found:', ', '.join(paths))\n    if keep_last is not None:\n        paths = sorted(paths)[-keep_last:]\n    print('Loading:', ', '.join(paths))\n\n    x_in = Input(shape=(maxlen,))\n    outputs = []\n\n    for i, path in enumerate(paths):\n        m = get_model(x_in)\n        m.load_weights(path)\n        outputs.append(m.output)\n\n    shape = outputs[0].get_shape().as_list()\n    x = Lambda(lambda x: K.mean(K.stack(x, axis=0), axis=0),\n               output_shape=lambda _: shape)(outputs)\n    model = Model(inputs=x_in, outputs=x)\n    return model\n    ","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"99e9faa54976b63ca7582c7e33f6746f469c3356"},"cell_type":"markdown","source":"**F1 score**"},{"metadata":{"trusted":true,"_uuid":"4dc13004dfd9ba1a81490ec347241acab90501d6"},"cell_type":"code","source":"# https://www.kaggle.com/hireme/fun-api-keras-f1-metric-cyclical-learning-rate/code\n\ndef f1(y_true, y_pred):\n    '''\n    metric from here \n    https://stackoverflow.com/questions/43547402/how-to-calculate-f1-macro-in-keras\n    '''\n    def recall(y_true, y_pred):\n        \"\"\"Recall metric.\n\n        Only computes a batch-wise average of recall.\n\n        Computes the recall, a metric for multi-label classification of\n        how many relevant items are selected.\n        \"\"\"\n        true_positives = K.sum(K.round(K.clip(y_true * y_pred, 0, 1)))\n        possible_positives = K.sum(K.round(K.clip(y_true, 0, 1)))\n        recall = true_positives / (possible_positives + K.epsilon())\n        return recall\n\n    def precision(y_true, y_pred):\n        \"\"\"Precision metric.\n\n        Only computes a batch-wise average of precision.\n\n        Computes the precision, a metric for multi-label classification of\n        how many selected items are relevant.\n        \"\"\"\n        true_positives = K.sum(K.round(K.clip(y_true * y_pred, 0, 1)))\n        predicted_positives = K.sum(K.round(K.clip(y_pred, 0, 1)))\n        precision = true_positives / (predicted_positives + K.epsilon())\n        return precision\n    precision = precision(y_true, y_pred)\n    recall = recall(y_true, y_pred)\n    return 2*((precision*recall)/(precision+recall+K.epsilon()))","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"5a676c3a275514a3351edf306e02d832a5f39317"},"cell_type":"markdown","source":"**Attention layer**"},{"metadata":{"trusted":true,"_uuid":"84e00df2c7b94205f5588af503f62412c48f46f3"},"cell_type":"code","source":"# https://www.kaggle.com/suicaokhoailang/lstm-attention-baseline-0-652-lb\n\nclass Attention(Layer):\n    def __init__(self, step_dim,\n                 W_regularizer=None, b_regularizer=None,\n                 W_constraint=None, b_constraint=None,\n                 bias=True, **kwargs):\n        self.supports_masking = True\n        self.init = initializers.get('glorot_uniform')\n\n        self.W_regularizer = regularizers.get(W_regularizer)\n        self.b_regularizer = regularizers.get(b_regularizer)\n\n        self.W_constraint = constraints.get(W_constraint)\n        self.b_constraint = constraints.get(b_constraint)\n\n        self.bias = bias\n        self.step_dim = step_dim\n        self.features_dim = 0\n        super(Attention, self).__init__(**kwargs)\n\n    def build(self, input_shape):\n        assert len(input_shape) == 3\n\n        self.W = self.add_weight((input_shape[-1],),\n                                 initializer=self.init,\n                                 name='{}_W'.format(self.name),\n                                 regularizer=self.W_regularizer,\n                                 constraint=self.W_constraint)\n        self.features_dim = input_shape[-1]\n\n        if self.bias:\n            self.b = self.add_weight((input_shape[1],),\n                                     initializer='zero',\n                                     name='{}_b'.format(self.name),\n                                     regularizer=self.b_regularizer,\n                                     constraint=self.b_constraint)\n        else:\n            self.b = None\n\n        self.built = True\n\n    def compute_mask(self, input, input_mask=None):\n        return None\n\n    def call(self, x, mask=None):\n        features_dim = self.features_dim\n        step_dim = self.step_dim\n\n        eij = K.reshape(K.dot(K.reshape(x, (-1, features_dim)),\n                        K.reshape(self.W, (features_dim, 1))), (-1, step_dim))\n\n        if self.bias:\n            eij += self.b\n\n        eij = K.tanh(eij)\n\n        a = K.exp(eij)\n\n        if mask is not None:\n            a *= K.cast(mask, K.floatx())\n\n        a /= K.cast(K.sum(a, axis=1, keepdims=True) + K.epsilon(), K.floatx())\n\n        a = K.expand_dims(a)\n        weighted_input = x * a\n        return K.sum(weighted_input, axis=1)\n\n    def compute_output_shape(self, input_shape):\n        return input_shape[0],  self.features_dim","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d96793d88c22274d985436e192f62970c227c324"},"cell_type":"markdown","source":"**LSTM models**"},{"metadata":{"trusted":true,"_uuid":"05164d541a0c35cae727d0338548d156efe21427"},"cell_type":"code","source":"def get_model(inp):\n    x = Embedding(max_features, embed_size, weights=[embedding_matrix], trainable=False)(inp)\n    \n    x = SpatialDropout1D(0.1)(x)\n    x = Bidirectional(CuDNNLSTM(40, return_sequences=True))(x)\n    y = Bidirectional(CuDNNGRU(40, return_sequences=True))(x)\n    \n    atten_1 = Attention(maxlen)(x) # skip connect\n    atten_2 = Attention(maxlen)(y)\n    avg_pool = GlobalAveragePooling1D()(y)\n    max_pool = GlobalMaxPooling1D()(y)\n    \n    conc = concatenate([atten_1, atten_2, avg_pool, max_pool])\n    conc = Dense(16, activation=\"relu\")(conc)\n    conc = Dropout(0.1)(conc)\n    outp = Dense(1, activation=\"sigmoid\")(conc)    \n\n    model = Model(inputs=inp, outputs=outp)\n    \n    return model","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"a8c857424e9c9f1703a71c1c0ade28713314dd29"},"cell_type":"markdown","source":"**Train and predict**"},{"metadata":{"trusted":true,"_uuid":"e8523d876b6eae762e673b777cc7af4d7f085792"},"cell_type":"code","source":"# https://www.kaggle.com/strideradu/word2vec-and-gensim-go-go-go\ndef train_pred(model, train_X, train_y, val_X, val_y, epochs=2):\n    cbs = [Snapshot('snapshots', nb_epochs=epochs, verbose=0, nb_cycles=3)]\n    model.compile(loss='binary_crossentropy', optimizer='adam', metrics=[f1])\n    model.fit(train_X, train_y, batch_size=512, epochs=epochs, validation_data=(val_X, val_y), callbacks=cbs, verbose=0)\n    del model    \n    \n    model = load_ensemble('snapshots', 2)\n    model.compile(loss='binary_crossentropy', optimizer='adam', metrics=[f1])\n        \n    pred_val_y = model.predict([val_X], batch_size=1024, verbose=0)\n\n    best_thresh = 0.5\n    best_score = 0.0\n    for thresh in np.arange(0.1, 0.501, 0.01):\n        thresh = np.round(thresh, 2)\n        score = metrics.f1_score(val_y, (pred_val_y > thresh).astype(int))\n        if score > best_score:\n            best_thresh = thresh\n            best_score = score\n\n    print(\"Val F1 Score: {:.4f}\".format(best_score))\n\n    pred_test_y = model.predict([test_X], batch_size=1024, verbose=0)\n    return pred_val_y, pred_test_y, best_score","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"f79081928ca032fbfe3b90c6d3ce91cf57d443d8"},"cell_type":"markdown","source":"**Main part: load, train, pred and blend**"},{"metadata":{"trusted":true,"_uuid":"99d03d2eb63600f1b222522616eab3fa35819f37"},"cell_type":"code","source":"train_X, test_X, train_y, word_index = load_and_prec()\nembedding_matrix_1 = load_glove(word_index)\n# embedding_matrix_2 = load_fasttext(word_index)\nembedding_matrix_3 = load_para(word_index)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f0ea9b1468bd7cd3ceead2593c641900dc3a2a77"},"cell_type":"code","source":"## Simple average: http://aclweb.org/anthology/N18-2031\n\n# We have presented an argument for averaging as\n# a valid meta-embedding technique, and found experimental\n# performance to be close to, or in some cases \n# better than that of concatenation, with the\n# additional benefit of reduced dimensionality  \n\n\n## Unweighted DME in https://arxiv.org/pdf/1804.07983.pdf\n\n# “The downside of concatenating embeddings and \n#  giving that as input to an RNN encoder, however,\n#  is that the network then quickly becomes inefficient\n#  as we combine more and more embeddings.”\n  \nembedding_matrix = np.mean([embedding_matrix_1, embedding_matrix_3], axis = 0)\nnp.shape(embedding_matrix)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"89b49b33d6204e1695888cf0a5fe683d4d37ef00"},"cell_type":"code","source":"# https://www.kaggle.com/ryanzhang/tfidf-naivebayes-logreg-baseline\n\ndef threshold_search(y_true, y_proba):\n    best_threshold = 0\n    best_score = 0\n    for threshold in [i * 0.01 for i in range(100)]:\n        score = f1_score(y_true=y_true, y_pred=y_proba > threshold)\n        if score > best_score:\n            best_threshold = threshold\n            best_score = score\n    search_result = {'threshold': best_threshold, 'f1': best_score}\n    return search_result","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"7aae47c7ded0b4f4849fe68b8c2f282ea03e1d20"},"cell_type":"code","source":"DATA_SPLIT_SEED = 2018\n\ntrain_meta = np.zeros(train_y.shape)\ntest_meta = np.zeros(test_X.shape[0])\nsplits = list(StratifiedKFold(n_splits=5, shuffle=True, random_state=DATA_SPLIT_SEED).split(train_X, train_y))\nfor idx, (train_idx, valid_idx) in enumerate(splits):\n        X_train = train_X[train_idx]\n        y_train = train_y[train_idx]\n        X_val = train_X[valid_idx]\n        y_val = train_y[valid_idx]\n        model = get_model(Input(shape=(maxlen,)))\n        pred_val_y, pred_test_y, best_score = train_pred(model, X_train, y_train, X_val, y_val, epochs = 6)\n        train_meta[valid_idx] = pred_val_y.reshape(-1)\n        test_meta += pred_test_y.reshape(-1) / len(splits)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d6f6012c790d09b4223d4c91ca4776b7815f8646"},"cell_type":"code","source":"search_result = threshold_search(train_y, train_meta)\nprint(search_result)\n\nsub = pd.read_csv('../input/sample_submission.csv')\nsub.prediction = test_meta > search_result['threshold']\nsub.to_csv(\"submission.csv\", index=False)","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}