{"cells":[{"metadata":{"_uuid":"18784d60d3b7bdc2cf24a296519e9a93cb0c61fb"},"cell_type":"markdown","source":"## General information\n\nIn this kernel I'll work with data from Quora Insincere Questions Classification Competition.\n\nThis dataset is interesting for NLP researching. We will try to find insincere questions which aren't usefull or are even harmful. I'll do a simple EDA and try an LSTM-CNN model. \n\n![](https://pbs.twimg.com/profile_images/1013607595616038912/pRq_huGc_400x400.jpg)"},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"_kg_hide-input":true},"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n%matplotlib inline\n\nfrom nltk.tokenize import TweetTokenizer\nimport datetime\nimport lightgbm as lgb\nfrom scipy import stats\nfrom scipy.sparse import hstack, csr_matrix\nfrom sklearn.model_selection import train_test_split, cross_val_score\nfrom wordcloud import WordCloud\nfrom collections import Counter\nfrom nltk.corpus import stopwords\nfrom nltk.util import ngrams\nfrom sklearn.feature_extraction.text import TfidfVectorizer\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.svm import LinearSVC\nfrom sklearn.multiclass import OneVsRestClassifier\nimport time\npd.set_option('max_colwidth',400)\n\nfrom keras.preprocessing.text import Tokenizer\nfrom keras.preprocessing.sequence import pad_sequences\nfrom keras.layers import Dense, Input, LSTM, Embedding, Dropout, Activation, Conv1D, GRU, CuDNNGRU, CuDNNLSTM, BatchNormalization\nfrom keras.layers import Bidirectional, GlobalMaxPool1D, MaxPooling1D, Add, Flatten, Masking\nfrom keras.layers import GlobalAveragePooling1D, GlobalMaxPooling1D, concatenate, SpatialDropout1D\nfrom keras.models import Model, load_model\nfrom keras import initializers, regularizers, constraints, optimizers, layers, callbacks\nfrom keras import backend as K\nfrom keras.engine import InputSpec, Layer\nfrom keras.optimizers import Adam\n\nfrom keras.callbacks import ModelCheckpoint, TensorBoard, Callback, EarlyStopping\nfrom sklearn.preprocessing import OneHotEncoder","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ab0fa74be8858a2b3197cf2761c85cc6964f5600"},"cell_type":"code","source":"import os\nprint(os.listdir(\"../input/embeddings/glove.840B.300d/\"))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"2b77f6a1c831c98851143feb25c9903cb1154bf2","_kg_hide-input":true},"cell_type":"code","source":"train = pd.read_csv(\"../input/train.csv\")\ntest = pd.read_csv(\"../input/test.csv\")\nsub = pd.read_csv('../input/sample_submission.csv')","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"28c5b2320f5ef863df3d0b4e4d9175c59bd61ef0"},"cell_type":"markdown","source":"## Data overview\n\nThis is a kernel competition, where we can't use external data. As a result we can use only train and test datasets as well as embeddings which were provided by organizers."},{"metadata":{"trusted":true,"_uuid":"1558909b0a5c120c1d5ddc5be4f5a952fcb4971e"},"cell_type":"code","source":"import os\nprint('Available embeddings:', os.listdir(\"../input/embeddings/\"))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"0c7299c8895405ef00049595ead1ef89649ba71b"},"cell_type":"code","source":"train[\"target\"].value_counts()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"cdbe2595e31608b72cfbdc8d4bfc75840bfe3a0d"},"cell_type":"markdown","source":"We have a seriuos disbalance - only ~6% of data are positive. No wonder the metric for the competition is f1-score."},{"metadata":{"trusted":true,"_uuid":"afaa845d44d72b9997ce037ab547ab4010701311"},"cell_type":"code","source":"train.head()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"0817c86624cc43ea1df0eba310ba41f4f799f8da"},"cell_type":"markdown","source":"In the dataset we have only texts of questions."},{"metadata":{"trusted":true,"_uuid":"54a553b7e92a2a0a3d491ccf92b437011b813c85"},"cell_type":"code","source":"print('Average word length of questions in train is {0:.0f}.'.format(np.mean(train['question_text'].apply(lambda x: len(x.split())))))\nprint('Average word length of questions in test is {0:.0f}.'.format(np.mean(test['question_text'].apply(lambda x: len(x.split())))))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"7861669f72f36145c25911926a51bc688f51d473"},"cell_type":"code","source":"print('Max word length of questions in train is {0:.0f}.'.format(np.max(train['question_text'].apply(lambda x: len(x.split())))))\nprint('Max word length of questions in test is {0:.0f}.'.format(np.max(test['question_text'].apply(lambda x: len(x.split())))))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"1b1303137eb44cc0de3329921751c48209037562"},"cell_type":"code","source":"print('Average character length of questions in train is {0:.0f}.'.format(np.mean(train['question_text'].apply(lambda x: len(x)))))\nprint('Average character length of questions in test is {0:.0f}.'.format(np.mean(test['question_text'].apply(lambda x: len(x)))))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b82f366ef8b46b0115e9940c7966849023545733"},"cell_type":"markdown","source":"As we can see on average questions in train and test datasets are similar, but there are quite long questions in train dataset."},{"metadata":{"trusted":true,"_uuid":"9d081b2f0d46faf01a943c309568c27f92462f94"},"cell_type":"code","source":"tk = Tokenizer(lower = True, filters='')\nfull_text = list(train['question_text'].values) + list(test['question_text'].values)\ntk.fit_on_texts(full_text)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"33929a60e1872b73e40424daa1178a3d8fbf8f5a"},"cell_type":"code","source":"train_tokenized = tk.texts_to_sequences(train['question_text'])\ntest_tokenized = tk.texts_to_sequences(test['question_text'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"4330f6c01064b5fda4ca9661dc4f1cefb439cf75"},"cell_type":"code","source":"train['question_text'].apply(lambda x: len(x.split())).plot(kind='hist');\nplt.yscale('log');\nplt.title('Distribution of question text length in characters')","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"05c7e2c63afb343b64835432721a42f81dd53626"},"cell_type":"markdown","source":"We can see that most of the questions are 40 words long or shorter. Let's try having sequence length equal to 70 for now."},{"metadata":{"trusted":true,"_uuid":"0a3f7fc48edb7d8d4aec66042dfb45e5af225c44"},"cell_type":"code","source":"max_len = 60\nX_train = pad_sequences(train_tokenized, maxlen = max_len)\nX_test = pad_sequences(test_tokenized, maxlen = max_len)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"744ee4a1fbc66cb47aae6a18f829dea284b860c9"},"cell_type":"code","source":"embedding_path = \"../input/embeddings/glove.840B.300d/glove.840B.300d.txt\"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"5cd86c2400db5ea1511b107250c8aa8c98ea909b"},"cell_type":"code","source":"embed_size = 300\nmax_features = 30000","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"2948d63cafb76b446c5e167a6dce7915ca43da6d"},"cell_type":"code","source":"def get_coefs(word,*arr): return word, np.asarray(arr, dtype='float32')\nembedding_index = dict(get_coefs(*o.strip().split(\" \")) for o in open(embedding_path, encoding='utf-8'))\n\nword_index = tk.word_index\nnb_words = min(max_features, len(word_index))\nembedding_matrix = np.zeros((nb_words + 1, embed_size))\nfor word, i in word_index.items():\n    if i >= max_features: continue\n    embedding_vector = embedding_index.get(word)\n    if embedding_vector is not None: embedding_matrix[i] = embedding_vector","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"9011f3f1e1affbe0795e2f7a98d4f0e98a8c4925"},"cell_type":"code","source":"ohe = OneHotEncoder(sparse=False)\ny_ohe = ohe.fit_transform(train['target'].values.reshape(-1, 1))","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"58c29867982a695f802700c77ebb213a52e18659"},"cell_type":"markdown","source":"For now I'll use an architecture from my previous [kernel](https://www.kaggle.com/artgor/movie-review-sentiment-analysis-eda-and-models) in another competition.\n\nThe architecture in the following:\n- input with embedding;\n- then we have separate \"branches\" - GRU and LSTM;\n- each \"branch\" is processed by two Conv1D layers separately;\n- each Conv1D layer has average and max pooling layers;\n- all pooling layers are concatenated;\n- two dense layers in the end;"},{"metadata":{"trusted":true,"_uuid":"55bc04ce8d311a8176886a211217d4e4459d4700"},"cell_type":"code","source":"def build_model(lr=0.0, lr_d=0.0, units=0, spatial_dr=0.0, kernel_size1=3, kernel_size2=2, dense_units=128, dr=0.1, conv_size=32, epochs=20):\n    file_path = \"best_model.hdf5\"\n    check_point = ModelCheckpoint(file_path, monitor = \"val_loss\", verbose = 1,\n                                  save_best_only = True, mode = \"min\")\n    early_stop = EarlyStopping(monitor = \"val_loss\", mode = \"min\", patience = 3)\n\n    inp = Input(shape = (max_len,))\n    x = Embedding(max_features + 1, embed_size, weights = [embedding_matrix], trainable = False)(inp)\n    x1 = SpatialDropout1D(spatial_dr)(x)\n\n    x_gru = Bidirectional(CuDNNGRU(units, return_sequences = True))(x1)\n    x_lstm = Bidirectional(CuDNNLSTM(units, return_sequences = True))(x1)\n    \n    x_conv1 = Conv1D(conv_size, kernel_size=kernel_size1, padding='valid', kernel_initializer='he_uniform')(x_gru)\n    avg_pool1_gru = GlobalAveragePooling1D()(x_conv1)\n    max_pool1_gru = GlobalMaxPooling1D()(x_conv1)\n    \n    x_conv2 = Conv1D(conv_size, kernel_size=kernel_size2, padding='valid', kernel_initializer='he_uniform')(x_gru)\n    avg_pool2_gru = GlobalAveragePooling1D()(x_conv2)\n    max_pool2_gru = GlobalMaxPooling1D()(x_conv2)\n    \n    \n    x_conv3 = Conv1D(conv_size, kernel_size=kernel_size1, padding='valid', kernel_initializer='he_uniform')(x_lstm)\n    avg_pool1_lstm = GlobalAveragePooling1D()(x_conv3)\n    max_pool1_lstm = GlobalMaxPooling1D()(x_conv3)\n    \n    x_conv4 = Conv1D(conv_size, kernel_size=kernel_size2, padding='valid', kernel_initializer='he_uniform')(x_lstm)\n    avg_pool2_lstm = GlobalAveragePooling1D()(x_conv4)\n    max_pool2_lstm = GlobalMaxPooling1D()(x_conv4)\n    \n    \n    x = concatenate([avg_pool1_gru, max_pool1_gru, avg_pool2_gru, max_pool2_gru,\n                    avg_pool1_lstm, max_pool1_lstm, avg_pool2_lstm, max_pool2_lstm])\n    x = BatchNormalization()(x)\n    x = Dropout(dr)(Dense(dense_units, activation='relu') (x))\n    x = BatchNormalization()(x)\n    x = Dropout(dr)(Dense(int(dense_units / 2), activation='relu') (x))\n    x = Dense(2, activation = \"sigmoid\")(x)\n    model = Model(inputs = inp, outputs = x)\n    model.compile(loss = \"binary_crossentropy\", optimizer = Adam(lr = lr, decay = lr_d), metrics = [\"accuracy\"])\n    model.summary()\n    history = model.fit(X_train, y_ohe, batch_size = 512, epochs = epochs, validation_split=0.1, \n                        verbose = 1, callbacks = [check_point, early_stop])\n    model = load_model(file_path)\n    return model","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"90ac92a570fe3c180252dfcf1f22c5f98a08e615"},"cell_type":"code","source":"#%%time\n#model = build_model(lr = 1e-4, lr_d = 0, units = 64, spatial_dr = 0.5, kernel_size1=4, kernel_size2=3, dense_units=16, dr=0.1, conv_size=16, epochs=5)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"8b206256803deeb5ba630578da1175b3cc01f555"},"cell_type":"code","source":"# pred = model.predict(X_test, batch_size = 1024, verbose = 1)\n# predictions = np.round(np.argmax(pred, axis=1)).astype(int)\n# sub['prediction'] = predictions\n# sub.to_csv(\"submission.csv\", index=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"754211d82f222f50cbb19712aca94984b2191588"},"cell_type":"code","source":"def build_model1(lr=0.0, lr_d=0.0, units=0, spatial_dr=0.0, kernel_size1=3, kernel_size2=2, dense_units=128, dr=0.1, conv_size=32, epochs=20):\n    file_path = \"best_model.hdf5\"\n    check_point = ModelCheckpoint(file_path, monitor = \"val_loss\", verbose = 1,\n                                  save_best_only = True, mode = \"min\")\n    early_stop = EarlyStopping(monitor = \"val_loss\", mode = \"min\", patience = 3)\n\n    inp = Input(shape = (max_len,))\n    x = Embedding(max_features + 1, embed_size, weights = [embedding_matrix], trainable = False)(inp)\n    x1 = SpatialDropout1D(spatial_dr)(x)\n\n    x_gru = Bidirectional(CuDNNGRU(units, return_sequences = True))(x1)\n    \n    x_conv1 = Conv1D(conv_size, kernel_size=kernel_size1, padding='valid', kernel_initializer='he_uniform')(x_gru)\n    avg_pool1_gru = GlobalAveragePooling1D()(x_conv1)\n    max_pool1_gru = GlobalMaxPooling1D()(x_conv1)\n    \n    x_conv2 = Conv1D(conv_size, kernel_size=kernel_size2, padding='valid', kernel_initializer='he_uniform')(x_gru)\n    avg_pool2_gru = GlobalAveragePooling1D()(x_conv2)\n    max_pool2_gru = GlobalMaxPooling1D()(x_conv2)\n\n    \n    \n    x = concatenate([avg_pool1_gru, max_pool1_gru, avg_pool2_gru, max_pool2_gru])\n    x = BatchNormalization()(x)\n    x = Dropout(dr)(Dense(dense_units, activation='relu') (x))\n    x = BatchNormalization()(x)\n    x = Dropout(dr)(Dense(int(dense_units / 2), activation='relu') (x))\n    x = Dense(2, activation = \"sigmoid\")(x)\n    model = Model(inputs = inp, outputs = x)\n    model.compile(loss = \"binary_crossentropy\", optimizer = Adam(lr = lr, decay = lr_d), metrics = [\"accuracy\"])\n    model.summary()\n    history = model.fit(X_train, y_ohe, batch_size = 512, epochs = epochs, validation_split=0.1, \n                        verbose = 1, callbacks = [check_point, early_stop])\n    model = load_model(file_path)\n    return model","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"552f593475f579d8c2822ee6c2d6b52ed1abc3e5"},"cell_type":"code","source":"#model1 = build_model1(lr = 1e-4, lr_d = 1e-7, units = 256, spatial_dr = 0.3, kernel_size1=4, kernel_size2=3, dense_units=32, dr=0.1, conv_size=32, epochs=5)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"0055ab6717740b4b234ada9aacb0c78074e64b55"},"cell_type":"code","source":"#model1_1 = build_model1(lr = 1e-4, lr_d = 1e-7, units = 128, spatial_dr = 0.3, kernel_size1=4, kernel_size2=3, dense_units=32, dr=0.1, conv_size=32, epochs=5)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"836e8eec34c09dcca660d35f39ddd5ec4801147a"},"cell_type":"code","source":"def build_model2(lr=0.0, lr_d=0.0, units=0, spatial_dr=0.0, kernel_size1=3, kernel_size2=2, dense_units=128, dr=0.1, conv_size=32, epochs=20):\n    file_path = \"best_model.hdf5\"\n    check_point = ModelCheckpoint(file_path, monitor = \"val_loss\", verbose = 1,\n                                  save_best_only = True, mode = \"min\")\n    early_stop = EarlyStopping(monitor = \"val_loss\", mode = \"min\", patience = 3)\n\n    inp = Input(shape = (max_len,))\n    x = Embedding(max_features + 1, embed_size, weights = [embedding_matrix], trainable = False)(inp)\n    x1 = SpatialDropout1D(spatial_dr)(x)\n\n    x_gru = Bidirectional(CuDNNGRU(units * 2, return_sequences = True))(x1)\n    x_gru = Bidirectional(CuDNNGRU(units, return_sequences = True))(x_gru)\n    \n    x_conv1 = Conv1D(conv_size, kernel_size=kernel_size1, padding='valid', kernel_initializer='he_uniform')(x_gru)\n    avg_pool1_gru = GlobalAveragePooling1D()(x_conv1)\n    max_pool1_gru = GlobalMaxPooling1D()(x_conv1)\n    \n    x_conv2 = Conv1D(conv_size, kernel_size=kernel_size2, padding='valid', kernel_initializer='he_uniform')(x_gru)\n    avg_pool2_gru = GlobalAveragePooling1D()(x_conv2)\n    max_pool2_gru = GlobalMaxPooling1D()(x_conv2)\n  \n    x = concatenate([avg_pool1_gru, max_pool1_gru, avg_pool2_gru, max_pool2_gru])\n    x = BatchNormalization()(x)\n    x = Dropout(dr)(Dense(dense_units, activation='relu') (x))\n    x = BatchNormalization()(x)\n    x = Dropout(dr)(Dense(int(dense_units / 2), activation='relu') (x))\n    x = Dense(2, activation = \"sigmoid\")(x)\n    model = Model(inputs = inp, outputs = x)\n    model.compile(loss = \"binary_crossentropy\", optimizer = Adam(lr = lr, decay = lr_d), metrics = [\"accuracy\"])\n    model.summary()\n    history = model.fit(X_train, y_ohe, batch_size = 512, epochs = epochs, validation_split=0.1, \n                        verbose = 1, callbacks = [check_point, early_stop])\n    model = load_model(file_path)\n    return model","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"0284b87509f5b94a6d03db579be74dad0bc8ff73"},"cell_type":"code","source":"#%%time\n#model2 = build_model2(lr = 1e-4, lr_d = 1e-7, units = 256, spatial_dr = 0.3, kernel_size1=4, kernel_size2=3, dense_units=32, dr=0.1, conv_size=32, epochs=5)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ac612d038ef7cf161c4e7c678751d29e4e2c7a82"},"cell_type":"code","source":"#model3 = build_model2(lr = 1e-3, lr_d = 1e-7, units = 256, spatial_dr = 0.3, kernel_size1=4, kernel_size2=3, dense_units=32, dr=0.1, conv_size=16, epochs=5)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"3f6b13c224c17bf70eac09f9bc3846d8332fa8d4"},"cell_type":"code","source":"%%time\n#model4 = build_model2(lr = 1e-4, lr_d = 1e-7, units = 128, spatial_dr = 0.3, kernel_size1=4, kernel_size2=3, dense_units=32, dr=0.1, conv_size=8, epochs=5)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"582cadcbd9fdd70e4c6da7160d743057fd9a487e"},"cell_type":"code","source":"#model5 = build_model2(lr = 1e-4, lr_d = 1e-7, units = 256, spatial_dr = 0.1, kernel_size1=4, kernel_size2=3, dense_units=32, dr=0.1, conv_size=16, epochs=5)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"04d92de6fc799aca62d3b41c588973d2480b2672"},"cell_type":"markdown","source":"### Model with attention\n\nhttps://github.com/Diyago/ML-DL-scripts/blob/9e161a96580efa9993805ca28f610df72fe36406/DEEP%20LEARNING/LSTM%20RNN/Sentiment%20analysis%20LSTM%20wth%20Bidirectional%20%20%2B%20Custom%20Attention.ipynb"},{"metadata":{"trusted":true,"_uuid":"0571d9edafb014eb480950a8c67fb892a4780240"},"cell_type":"code","source":"class Attention(Layer):\n    def __init__(self, step_dim,\n                 W_regularizer=None, b_regularizer=None,\n                 W_constraint=None, b_constraint=None,\n                 bias=True, **kwargs):\n        \"\"\"\n        Keras Layer that implements an Attention mechanism for temporal data.\n        Supports Masking.\n        Follows the work of Raffel et al. [https://arxiv.org/abs/1512.08756]\n        # Input shape\n            3D tensor with shape: `(samples, steps, features)`.\n        # Output shape\n            2D tensor with shape: `(samples, features)`.\n        :param kwargs:\n        Just put it on top of an RNN Layer (GRU/LSTM/SimpleRNN) with return_sequences=True.\n        The dimensions are inferred based on the output shape of the RNN.\n        Example:\n            model.add(LSTM(64, return_sequences=True))\n            model.add(Attention())\n        \"\"\"\n        self.supports_masking = True\n        #self.init = initializations.get('glorot_uniform')\n        self.init = initializers.get('glorot_uniform')\n\n        self.W_regularizer = regularizers.get(W_regularizer)\n        self.b_regularizer = regularizers.get(b_regularizer)\n\n        self.W_constraint = constraints.get(W_constraint)\n        self.b_constraint = constraints.get(b_constraint)\n\n        self.bias = bias\n        self.step_dim = step_dim\n        self.features_dim = 0\n        super(Attention, self).__init__(**kwargs)\n\n    def build(self, input_shape):\n        assert len(input_shape) == 3\n\n        self.W = self.add_weight((input_shape[-1],),\n                                 initializer=self.init,\n                                 name='{}_W'.format(self.name),\n                                 regularizer=self.W_regularizer,\n                                 constraint=self.W_constraint)\n        self.features_dim = input_shape[-1]\n\n        if self.bias:\n            self.b = self.add_weight((input_shape[1],),\n                                     initializer='zero',\n                                     name='{}_b'.format(self.name),\n                                     regularizer=self.b_regularizer,\n                                     constraint=self.b_constraint)\n        else:\n            self.b = None\n\n        self.built = True\n\n    def compute_mask(self, input, input_mask=None):\n        # do not pass the mask to the next layers\n        return None\n\n    def call(self, x, mask=None):\n        # eij = K.dot(x, self.W) TF backend doesn't support it\n\n        # features_dim = self.W.shape[0]\n        # step_dim = x._keras_shape[1]\n\n        features_dim = self.features_dim\n        step_dim = self.step_dim\n\n        eij = K.reshape(K.dot(K.reshape(x, (-1, features_dim)), K.reshape(self.W, (features_dim, 1))), (-1, step_dim))\n\n        if self.bias:\n            eij += self.b\n\n        eij = K.tanh(eij)\n\n        a = K.exp(eij)\n\n        # apply mask after the exp. will be re-normalized next\n        if mask is not None:\n            # Cast the mask to floatX to avoid float64 upcasting in theano\n            a *= K.cast(mask, K.floatx())\n\n        # in some cases especially in the early stages of training the sum may be almost zero\n        a /= K.cast(K.sum(a, axis=1, keepdims=True) + K.epsilon(), K.floatx())\n\n        a = K.expand_dims(a)\n        weighted_input = x * a\n    #print weigthted_input.shape\n        return K.sum(weighted_input, axis=1)\n\n    def compute_output_shape(self, input_shape):\n        #return input_shape[0], input_shape[-1]\n        return input_shape[0],  self.features_dim","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"143768540a1c3ad0783e49362e9018b7d46a1f07"},"cell_type":"markdown","source":"### Attention"},{"metadata":{"trusted":true,"_uuid":"19030bd2b6277b9e5dbc6933ca0abcf68fc7b3f9"},"cell_type":"code","source":"def build_model3(lr=0.0, lr_d=0.0, units=0, spatial_dr=0.0, dense_units=128, dr=0.1, use_attention=True):\n    inp = Input(shape = (max_len,))\n    x = Embedding(max_features + 1, embed_size, weights = [embedding_matrix], trainable = False)(inp)\n    x1 = SpatialDropout1D(spatial_dr)(x)\n\n    x_gru = Bidirectional(CuDNNGRU(units * 2, return_sequences = True))(x1)\n    if use_attention:\n        x_att = Attention(max_len)(x_gru)\n        x = Dropout(dr)(Dense(dense_units, activation='relu') (x_att))\n    else:\n        x_att = Flatten() (x_gru)\n        x = Dropout(dr)(Dense(dense_units, activation='relu') (x_att))\n\n    x = BatchNormalization()(x)\n    #x = Dropout(dr)(Dense(int(dense_units / 2), activation='relu') (x))\n    x = Dense(2, activation = \"sigmoid\")(x)\n    model = Model(inputs = inp, outputs = x)\n    model.compile(loss = \"binary_crossentropy\", optimizer = Adam(lr = lr, decay = lr_d), metrics = [\"accuracy\"])\n    #model.summary()\n    #history = model.fit(X_train, y_ohe, batch_size = 512, epochs = epochs, validation_split=0.1, \n    #                    verbose = 1, callbacks = [check_point, early_stop])\n    #model = load_model(file_path)\n    return model","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"1e7072209b383437e44560d062c1614418a1607a","collapsed":true},"cell_type":"code","source":"# %%time\n# file_path = \"best_model.hdf5\"\n# check_point = ModelCheckpoint(file_path, monitor = \"val_loss\", verbose = 1,\n#                               save_best_only = True, mode = \"min\")\n# early_stop = EarlyStopping(monitor = \"val_loss\", mode = \"min\", patience = 3)\n# model6 = build_model3(lr = 1e-3, lr_d = 1e-7, units = 64, spatial_dr = 0.3, dense_units=16, dr=0.1, use_attention=True)\n# history = model6.fit(X_train, y_ohe, batch_size = 512, epochs = 10, validation_split=0.1, \n#                     verbose = 1, callbacks = [check_point, early_stop])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"4c0cab2368cf13a33e56a92eec20ac01254c3529"},"cell_type":"code","source":"# #%%time\n# file_path = \"best_model.hdf5\"\n# check_point = ModelCheckpoint(file_path, monitor = \"val_loss\", verbose = 1,\n#                               save_best_only = True, mode = \"min\")\n# early_stop = EarlyStopping(monitor = \"val_loss\", mode = \"min\", patience = 3)\n# model6_1 = build_model3(lr = 1e-3, lr_d = 1e-7, units = 64, spatial_dr = 0.3, dense_units=16, dr=0.1, use_attention=False)\n# history = model6_1.fit(X_train, y_ohe, batch_size = 512, epochs = 5, validation_split=0.1, \n#                     verbose = 1, callbacks = [check_point, early_stop])","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"bbc608b1a70e5bc21df515136b6225da29e797a9"},"cell_type":"markdown","source":"### One branch"},{"metadata":{"trusted":true,"_uuid":"fecf3405fccd307de507cea500df6b559e70cc95"},"cell_type":"code","source":"def build_model4(lr=0.0, lr_d=0.0, units=0, spatial_dr=0.0, kernel_size1=3, kernel_size2=2, dense_units=128, dr=0.1, conv_size=32, epochs=20):\n    file_path = \"best_model.hdf5\"\n    check_point = ModelCheckpoint(file_path, monitor = \"val_loss\", verbose = 1,\n                                  save_best_only = True, mode = \"min\")\n    early_stop = EarlyStopping(monitor = \"val_loss\", mode = \"min\", patience = 3)\n\n    inp = Input(shape = (max_len,))\n    x = Embedding(max_features + 1, embed_size, weights = [embedding_matrix], trainable = False)(inp)\n    x1 = SpatialDropout1D(spatial_dr)(x)\n\n    x_gru = Bidirectional(CuDNNGRU(units, return_sequences = True))(x1)\n    \n    x_conv1 = Conv1D(conv_size, kernel_size=kernel_size1, padding='valid', kernel_initializer='he_uniform')(x_gru)\n    avg_pool1_gru = GlobalAveragePooling1D()(x_conv1)\n    max_pool1_gru = GlobalMaxPooling1D()(x_conv1)\n       \n    x = concatenate([avg_pool1_gru, max_pool1_gru])\n    x = BatchNormalization()(x)\n    x = Dropout(dr)(Dense(dense_units, activation='relu') (x))\n    x = BatchNormalization()(x)\n    #x = Dropout(dr)(Dense(int(dense_units / 2), activation='relu') (x))\n    x = Dense(2, activation = \"sigmoid\")(x)\n    model = Model(inputs = inp, outputs = x)\n    model.compile(loss = \"binary_crossentropy\", optimizer = Adam(lr = lr, decay = lr_d), metrics = [\"accuracy\"])\n    model.summary()\n    history = model.fit(X_train, y_ohe, batch_size = 512, epochs = epochs, validation_split=0.1, \n                        verbose = 1, callbacks = [check_point, early_stop])\n    model = load_model(file_path)\n    return model","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"29fcf5b25f20763deb75612eac29d58add377fee"},"cell_type":"code","source":"#%%time\n#model7 = build_model4(lr = 1e-4, lr_d = 1e-7, units = 64, spatial_dr = 0.3, kernel_size1=3, dense_units=32, dr=0.1, conv_size=8, epochs=5)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"1b21c1b068660aaca49e9f1d1c8259bd5c97ad56"},"cell_type":"code","source":"#model8 = build_model4(lr = 1e-4, lr_d = 1e-7, units = 128, spatial_dr = 0.3, kernel_size1=4, dense_units=32, dr=0.1, conv_size=8, epochs=5)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"f12bb4ca1ee7bca3caa41426590330365e057d34"},"cell_type":"markdown","source":"### Masking"},{"metadata":{"trusted":true,"_uuid":"dcae5f568eac0945c349eb2253df8010d03faa97"},"cell_type":"code","source":"def build_model5(lr=0.0, lr_d=0.0, units=0, spatial_dr=0.0, kernel_size1=3, kernel_size2=2, dense_units=128, dr=0.1, conv_size=32, epochs=20):\n    file_path = \"best_model.hdf5\"\n    check_point = ModelCheckpoint(file_path, monitor = \"val_loss\", verbose = 1,\n                                  save_best_only = True, mode = \"min\")\n    early_stop = EarlyStopping(monitor = \"val_loss\", mode = \"min\", patience = 3)\n\n    inp = Input(shape = (max_len,))\n    x = Embedding(max_features + 1, embed_size, weights = [embedding_matrix], trainable = False)(inp)\n    x1 = SpatialDropout1D(spatial_dr)(x)\n    x_m = Masking()(x1)\n    x_gru = LSTM(units)(x_m)\n\n    x = BatchNormalization()(x_gru)\n    x = Dropout(dr)(Dense(dense_units, activation='relu') (x))\n    x = BatchNormalization()(x)\n    #x = Dropout(dr)(Dense(int(dense_units / 2), activation='relu') (x))\n    x = Dense(2, activation = \"sigmoid\")(x)\n    model = Model(inputs = inp, outputs = x)\n    model.compile(loss = \"binary_crossentropy\", optimizer = Adam(lr = lr, decay = lr_d), metrics = [\"accuracy\"])\n    model.summary()\n    history = model.fit(X_train, y_ohe, batch_size = 512, epochs = epochs, validation_split=0.1, \n                        verbose = 1, callbacks = [check_point, early_stop])\n    model = load_model(file_path)\n    return model","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"3b98e6c57a2f16b29b8ae7265919eb309e36e770"},"cell_type":"code","source":"model9 = build_model5(lr = 1e-4, lr_d = 1e-7, units = 128, spatial_dr = 0.3, kernel_size1=4, dense_units=32, dr=0.1, conv_size=8, epochs=5)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e4c4dfb35bb2fa517a3cc3aa51fd663e14782ac0"},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"771278144d9fe02f509ee4e08b7a97d3e69b8283"},"cell_type":"code","source":"# pred1 = model.predict(X_test, batch_size = 1024, verbose = 1)\n# pred = pred1\n# pred2 = model2.predict(X_test, batch_size = 1024, verbose = 1)\n# pred += pred2\n# pred3 = model4.predict(X_test, batch_size = 1024, verbose = 1)\n# pred = pred3\n# pred4 = model6.predict(X_test, batch_size = 1024, verbose = 1)\n# pred += pred4\n# pred5 = model7.predict(X_test, batch_size = 1024, verbose = 1)\n# pred += pred5\n# pred = pred / 3","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"2c04e298f5c96a9561450874a0238f408984c39e"},"cell_type":"code","source":"pred = model9.predict(X_test, batch_size = 1024, verbose = 1)\n\npredictions = np.round(np.argmax(pred, axis=1)).astype(int)\nsub['prediction'] = predictions\nsub.to_csv(\"submission.csv\", index=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a01a26918d26e749d9ac454e68cb71830864236f"},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}