{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nimport os\nprint(os.listdir(\"../input\"))\n\n# Any results you write to the current directory are saved as output.","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"import itertools\nfrom tqdm import tqdm\nfrom sklearn.model_selection import train_test_split, StratifiedKFold\nfrom sklearn.metrics import f1_score, roc_curve, auc, confusion_matrix\n\nfrom keras.preprocessing.text import Tokenizer\nfrom keras.preprocessing.sequence import pad_sequences\nfrom keras.layers import (\n    Dense, Input, CuDNNLSTM, Embedding, Dropout, Activation, CuDNNGRU, \n    Conv1D ,GlobalMaxPool1D, GlobalMaxPooling1D, GlobalAveragePooling1D,\n    Conv2D, MaxPool2D, concatenate,\n    Reshape, Flatten, Concatenate, Dropout, SpatialDropout1D, Bidirectional, \n)\nfrom keras.layers.normalization import BatchNormalization\nfrom keras.optimizers import Adam\nfrom keras.models import Model\nfrom keras.engine.topology import Layer\nfrom keras import metrics\nfrom keras import initializers, regularizers, constraints, optimizers, layers\nfrom keras import backend as K\n\nimport matplotlib.pyplot as plt\n%matplotlib inline","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"1232cdbbf2d2c04663a84ce07f08b9fa6f006318"},"cell_type":"code","source":"DEBUG = False\nEMBED_SIZE = 300\nMAX_FEATURES = 10000 if DEBUG else 90000\nSEQUENCE_LENGTH = 50\nGLOVE = '../input/embeddings/glove.840B.300d/glove.840B.300d.txt'\nWIKI = '../input/embeddings/wiki-news-300d-1M/wiki-news-300d-1M.vec'\nPARAGRAM = '../input/embeddings/paragram_300_sl999/paragram_300_sl999.txt'","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"25195605991b8328f79685a52e4d19fd1083798f"},"cell_type":"code","source":"train_df = pd.read_csv('../input/train.csv')\ntest_df = pd.read_csv('../input/test.csv')\nif DEBUG:\n    train_df = train_df[:100000]\ntrain_df['question_text'] = train_df['question_text'].str.lower()\ntest_df['question_text'] = test_df['question_text'].str.lower()\nDEBUG, train_df.shape, test_df.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"423cd3b274285d9bf47b08803bc2f3cb99d667d1"},"cell_type":"code","source":"def load_embedding(file, tokenizer):\n    def split(line):\n        word, *arr = line.split(' ')\n        return word, np.asarray(arr, dtype='float32')\n    with open(file, encoding='utf8', errors='ignore') as f:\n        if DEBUG:\n            embeddings = dict(split(line) for line in tqdm(itertools.islice(f, 10000)) if len(line) > 100)\n        else:\n            embeddings = dict(split(line) for line in tqdm(f) if len(line) > 100)\n    values = np.stack(embeddings.values())\n    mean = values.mean()\n    std = values.std()\n    n_words = min(MAX_FEATURES, len(tokenizer.word_index))\n    \n    embedding_matrix = np.random.normal(mean, std, (n_words, EMBED_SIZE))\n    for word, i in tokenizer.word_index.items():\n        if i < MAX_FEATURES and word in embeddings:\n            embedding_matrix[i] = embeddings[word]\n            \n    return embedding_matrix","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"03c3b9982b24bc971bedd328ea1f91aea20f66b5"},"cell_type":"code","source":"# https://www.kaggle.com/suicaokhoailang/lstm-attention-baseline-0-652-lb\nclass Attention(Layer):\n    def __init__(self, step_dim,\n                 W_regularizer=None, b_regularizer=None,\n                 W_constraint=None, b_constraint=None,\n                 bias=True, **kwargs):\n        self.supports_masking = True\n        self.init = initializers.get('glorot_uniform')\n\n        self.W_regularizer = regularizers.get(W_regularizer)\n        self.b_regularizer = regularizers.get(b_regularizer)\n\n        self.W_constraint = constraints.get(W_constraint)\n        self.b_constraint = constraints.get(b_constraint)\n\n        self.bias = bias\n        self.step_dim = step_dim\n        self.features_dim = 0\n        super(Attention, self).__init__(**kwargs)\n\n    def build(self, input_shape):\n        assert len(input_shape) == 3\n\n        self.W = self.add_weight((input_shape[-1],),\n                                 initializer=self.init,\n                                 name='{}_W'.format(self.name),\n                                 regularizer=self.W_regularizer,\n                                 constraint=self.W_constraint)\n        self.features_dim = input_shape[-1]\n\n        if self.bias:\n            self.b = self.add_weight((input_shape[1],),\n                                     initializer='zero',\n                                     name='{}_b'.format(self.name),\n                                     regularizer=self.b_regularizer,\n                                     constraint=self.b_constraint)\n        else:\n            self.b = None\n\n        self.built = True\n\n    def compute_mask(self, input, input_mask=None):\n        return None\n\n    def call(self, x, mask=None):\n        features_dim = self.features_dim\n        step_dim = self.step_dim\n\n        eij = K.reshape(K.dot(K.reshape(x, (-1, features_dim)),\n                        K.reshape(self.W, (features_dim, 1))), (-1, step_dim))\n        if self.bias:\n            eij += self.b\n        eij = K.tanh(eij)\n        a = K.exp(eij)\n        \n        if mask is not None:\n            a *= K.cast(mask, K.floatx())\n\n        a /= K.cast(K.sum(a, axis=1, keepdims=True) + K.epsilon(), K.floatx())\n        a = K.expand_dims(a)\n        weighted_input = x * a\n        return K.sum(weighted_input, axis=1)\n\n    def compute_output_shape(self, input_shape):\n        return input_shape[0],  self.features_dim","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ea3e89d831b447446a8a2154d12105161f71d3e8"},"cell_type":"code","source":"def model_gru_atten_3(embedding):\n    input_ = Input(shape=(SEQUENCE_LENGTH,))\n    x = Embedding(MAX_FEATURES, EMBED_SIZE, weights=[embedding])(input_)\n    x = Bidirectional(CuDNNGRU(128, return_sequences=True))(x)\n    x = Bidirectional(CuDNNGRU(96, return_sequences=True))(x)\n    x = Bidirectional(CuDNNGRU(64, return_sequences=True))(x)\n    x = Attention(SEQUENCE_LENGTH)(x)\n    x = Dense(1, activation='sigmoid')(x)\n    model = Model(inputs=input_, outputs=x)\n    model.compile(loss='binary_crossentropy', optimizer='adam', metrics=['accuracy'])\n    \n    return model","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"1772db288ab467d8704808795ab2a0fed5993201"},"cell_type":"code","source":"class AveragedModel:\n    def __init__(self, tokenizer):\n        self.embeddings = [\n            load_embedding(GLOVE, tokenizer),\n            load_embedding(WIKI, tokenizer),\n            load_embedding(PARAGRAM, tokenizer),\n        ]\n        self.models = [model_gru_atten_3(embedding) for embedding in self.embeddings]\n        \n    def fit(self, *args, **kwargs):\n        for model in self.models:\n            model.fit(*args, **kwargs)\n            \n    def predict(self, *args, **kwargs):\n        outputs = [model.predict(*args, **kwargs) for model in self.models]\n        return np.array(outputs).mean(axis=0)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"8e409b00d1d159a1e5faa5cb9de9a7ead89b29a7"},"cell_type":"code","source":"def plot_base():\n    plt.plot([0, 1], [0, 1], color='navy', linestyle='--')\n    plt.xlim([0.0, 1.0])\n    plt.ylim([0.0, 1.05])\n    plt.xlabel('False Positive Rate')\n    plt.ylabel('True Positive Rate')\n    plt.title('Receiver operating characteristic')\n    plt.legend(loc=\"lower right\")\n    plt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"fb5aa4df482790d76a7ff79093c78344086da148"},"cell_type":"code","source":"import random\ndef drop_randomly(sequences, ratio=.1):\n    for sequence in sequences:\n        if random.random() < ratio and sequence:\n            del sequence[random.randint(0, len(sequence) - 1)]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e5f1842ea1aa4aa7046fc91f4d83c86ccb3fcc52"},"cell_type":"code","source":"X = train_df[\"question_text\"]\ny = train_df['target'].values\npredicted_y = np.zeros_like(y, dtype='float32')\n\nindices = []\nsplitted_Xy = []\nmodels = []\n\nsplitter = StratifiedKFold(2, random_state=0)\n\n\nfor train_index, test_index in splitter.split(X, y):\n    indices.append((train_index, test_index))\n    \n    tokenizer = Tokenizer(num_words=MAX_FEATURES)\n    tokenizer.fit_on_texts(X[train_index])\n    \n    train_X = tokenizer.texts_to_sequences(X[train_index])\n    drop_randomly(train_X, ratio=.1) # test for robustness\n    train_X = pad_sequences(train_X, maxlen=SEQUENCE_LENGTH)\n    test_X = tokenizer.texts_to_sequences(X[test_index])\n    test_X = pad_sequences(test_X, maxlen=SEQUENCE_LENGTH)\n    train_y = y[train_index]\n    test_y = y[test_index]\n\n    splitted_Xy.append((train_X, test_X, train_y, test_y))\n    model = AveragedModel(tokenizer)\n    models.append(model)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"037b9a201ec927cfb6e7f1c3c662eabfc0a4c28b"},"cell_type":"code","source":"for model, (train_X, test_X, train_y, test_y), (train_index, test_index) in zip(models, splitted_Xy, indices):\n    model.fit(train_X, train_y, batch_size=512, epochs=1, verbose=0)\n    \n    yp = model.predict([test_X], batch_size=512, verbose=0).flatten()\n    predicted_y[test_index] = yp\n    \n    \nfpr, tpr, thresholds = roc_curve(y, predicted_y, pos_label=1)\nroc_auc = auc(fpr, tpr)\nplt.plot(fpr, tpr, color='darkorange', label='ROC curve (area = %0.3f)' % roc_auc)\nplot_base()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"7fb04d430d129e6fcf32014e7552d4a8fc490698"},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}