{"cells":[{"metadata":{"_uuid":"18784d60d3b7bdc2cf24a296519e9a93cb0c61fb"},"cell_type":"markdown","source":"## General information\n\nIn this kernel I'll work with data from Quora Insincere Questions Classification Competition.\n\nThis dataset is interesting for NLP researching. We will try to find insincere questions which aren't usefull or are even harmful. I'll do a simple EDA and try an LSTM-CNN model. \n\n![](https://pbs.twimg.com/profile_images/1013607595616038912/pRq_huGc_400x400.jpg)"},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"_kg_hide-input":true},"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n%matplotlib inline\n\nfrom nltk.tokenize import TweetTokenizer\nimport datetime\nimport lightgbm as lgb\nfrom scipy import stats\nfrom scipy.sparse import hstack, csr_matrix\nfrom sklearn.model_selection import train_test_split, cross_val_score\nfrom wordcloud import WordCloud\nfrom collections import Counter\nfrom nltk.corpus import stopwords\nfrom nltk.util import ngrams\nfrom sklearn.feature_extraction.text import TfidfVectorizer\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.svm import LinearSVC\nfrom sklearn.multiclass import OneVsRestClassifier\nimport time\npd.set_option('max_colwidth',400)\n\nfrom keras.preprocessing.text import Tokenizer\nfrom keras.preprocessing.sequence import pad_sequences\nfrom keras.layers import Dense, Input, LSTM, Embedding, Dropout, Activation, Conv1D, GRU, CuDNNGRU, CuDNNLSTM, BatchNormalization\nfrom keras.layers import Bidirectional, GlobalMaxPool1D, MaxPooling1D, Add, Flatten, Masking\nfrom keras.layers import GlobalAveragePooling1D, GlobalMaxPooling1D, concatenate, SpatialDropout1D\nfrom keras.models import Model, load_model\nfrom keras import initializers, regularizers, constraints, optimizers, layers, callbacks\nfrom keras import backend as K\nfrom keras.engine import InputSpec, Layer\nfrom keras.optimizers import Adam\n\nfrom keras.callbacks import ModelCheckpoint, TensorBoard, Callback, EarlyStopping\nfrom sklearn.preprocessing import OneHotEncoder","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ab0fa74be8858a2b3197cf2761c85cc6964f5600"},"cell_type":"code","source":"import os\nprint(os.listdir(\"../input/embeddings/glove.840B.300d/\")) # 使用的是Glove_840B_300d，那麼EMBEDDING_DIM=300","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"2b77f6a1c831c98851143feb25c9903cb1154bf2","_kg_hide-input":true},"cell_type":"code","source":"train = pd.read_csv(\"../input/train.csv\")\ntest = pd.read_csv(\"../input/test.csv\")\nsub = pd.read_csv('../input/sample_submission.csv')","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"28c5b2320f5ef863df3d0b4e4d9175c59bd61ef0"},"cell_type":"markdown","source":"## Data overview\n\nThis is a kernel competition, where we can't use external data. As a result we can use only train and test datasets as well as embeddings which were provided by organizers."},{"metadata":{"trusted":true,"_uuid":"1558909b0a5c120c1d5ddc5be4f5a952fcb4971e"},"cell_type":"code","source":"import os\nprint('Available embeddings:', os.listdir(\"../input/embeddings/\"))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"0c7299c8895405ef00049595ead1ef89649ba71b"},"cell_type":"code","source":"train[\"target\"].value_counts()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"cdbe2595e31608b72cfbdc8d4bfc75840bfe3a0d"},"cell_type":"markdown","source":"We have a seriuos disbalance - only ~6% of data are positive. No wonder the metric for the competition is f1-score."},{"metadata":{"trusted":true,"_uuid":"afaa845d44d72b9997ce037ab547ab4010701311"},"cell_type":"code","source":"train.head()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"0817c86624cc43ea1df0eba310ba41f4f799f8da"},"cell_type":"markdown","source":"In the dataset we have only texts of questions."},{"metadata":{"trusted":true,"_uuid":"54a553b7e92a2a0a3d491ccf92b437011b813c85"},"cell_type":"code","source":"print('Average word length of questions in train is {0:.0f}.'.format(np.mean(train['question_text'].apply(lambda x: len(x.split())))))\nprint('Average word length of questions in test is {0:.0f}.'.format(np.mean(test['question_text'].apply(lambda x: len(x.split())))))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"7861669f72f36145c25911926a51bc688f51d473"},"cell_type":"code","source":"print('Max word length of questions in train is {0:.0f}.'.format(np.max(train['question_text'].apply(lambda x: len(x.split())))))\nprint('Max word length of questions in test is {0:.0f}.'.format(np.max(test['question_text'].apply(lambda x: len(x.split())))))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"1b1303137eb44cc0de3329921751c48209037562"},"cell_type":"code","source":"print('Average character length of questions in train is {0:.0f}.'.format(np.mean(train['question_text'].apply(lambda x: len(x)))))\nprint('Average character length of questions in test is {0:.0f}.'.format(np.mean(test['question_text'].apply(lambda x: len(x)))))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b82f366ef8b46b0115e9940c7966849023545733"},"cell_type":"markdown","source":"As we can see on average questions in train and test datasets are similar, but there are quite long questions in train dataset."},{"metadata":{"trusted":true,"_uuid":"9d081b2f0d46faf01a943c309568c27f92462f94"},"cell_type":"code","source":"max_features = 50000\ntk = Tokenizer(lower = True, filters='', num_words=max_features) # 序列化文本\nfull_text = list(train['question_text'].values) + list(test['question_text'].values) # 存储列表\ntk.fit_on_texts(full_text) # 要用以训练的文本列表","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c2a11e437f76d342d63643a72fc0e14ca2b79233"},"cell_type":"code","source":"full_text[:3], len(full_text)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"4eb72baaef227f99f4785f7bfd204744fff636ad"},"cell_type":"code","source":"train.question_text.describe()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"33929a60e1872b73e40424daa1178a3d8fbf8f5a"},"cell_type":"code","source":"# 输入：待转为序列的文本列表\n# 返回值：序列的列表，列表中每个序列对应于一段输入文本\ntrain_tokenized = tk.texts_to_sequences(train['question_text'].fillna('missing'))\ntest_tokenized = tk.texts_to_sequences(test['question_text'].fillna('missing')) # tk.texts_to_sequences方法会丢失train中未出现过的单词","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"675cc5b293ad2d632d5041f7d2eb797ff0d7b786"},"cell_type":"code","source":"'missing' in test_tokenized","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b61da335025486edc847e6704afd432c2ce5ba76"},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"4330f6c01064b5fda4ca9661dc4f1cefb439cf75"},"cell_type":"code","source":"train['question_text'].apply(lambda x: len(x.split())).plot(kind='hist');\nplt.yscale('log');\nplt.title('Distribution of question text length in characters')","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"05c7e2c63afb343b64835432721a42f81dd53626"},"cell_type":"markdown","source":"We can see that most of the questions are 40 words long or shorter. Let's try having sequence length equal to 70 for now."},{"metadata":{"trusted":true,"_uuid":"0a3f7fc48edb7d8d4aec66042dfb45e5af225c44"},"cell_type":"code","source":"max_len = 70\nX_train = pad_sequences(train_tokenized, maxlen = max_len)\nX_test = pad_sequences(test_tokenized, maxlen = max_len)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"744ee4a1fbc66cb47aae6a18f829dea284b860c9"},"cell_type":"code","source":"embedding_path = \"../input/embeddings/glove.840B.300d/glove.840B.300d.txt\"  # glove词向量\n#embedding_path = \"../input/embeddings/paragram_300_sl999/paragram_300_sl999.txt\"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"5cd86c2400db5ea1511b107250c8aa8c98ea909b"},"cell_type":"code","source":"embed_size = 300","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"2948d63cafb76b446c5e167a6dce7915ca43da6d"},"cell_type":"code","source":"def get_coefs(word,*arr):\n    return word, np.asarray(arr, dtype='float32') # 字符串列表转化\n\nembedding_index = dict(get_coefs(*o.strip().split(\" \")) for o in open(embedding_path, encoding='utf-8', errors='ignore'))\n\nword_index = tk.word_index # 字典，将单词（字符串）映射为它们的排名或者索引\nnb_words = min(max_features, len(word_index)) # max_features为最大单词数、len(word_index)为实际的单词的最大序号\nembedding_matrix = np.zeros((nb_words + 1, embed_size)) # keras需要预留一个全0层，所以要加1\n\nfor word, i in word_index.items():\n    if i >= max_features:\n        continue\n    embedding_vector = embedding_index.get(word)\n    if embedding_vector is not None:\n        embedding_matrix[i] = embedding_vector","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"9011f3f1e1affbe0795e2f7a98d4f0e98a8c4925"},"cell_type":"code","source":"ohe = OneHotEncoder(sparse=False)\ny_ohe = ohe.fit_transform(train['target'].values.reshape(-1, 1))","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"58c29867982a695f802700c77ebb213a52e18659"},"cell_type":"markdown","source":"For now I'll use an architecture from my previous [kernel](https://www.kaggle.com/artgor/movie-review-sentiment-analysis-eda-and-models) in another competition.\n\nThe architecture in the following:\n- input with embedding;\n- then we have separate \"branches\" - GRU and LSTM;\n- each \"branch\" is processed by two Conv1D layers separately;\n- each Conv1D layer has average and max pooling layers;\n- all pooling layers are concatenated;\n- two dense layers in the end;"},{"metadata":{"trusted":true,"_uuid":"55bc04ce8d311a8176886a211217d4e4459d4700"},"cell_type":"code","source":"def build_model(lr=0.0, lr_d=0.0, units=0, spatial_dr=0.0, kernel_size1=3, kernel_size2=2, dense_units=128, dr=0.1, conv_size=32, epochs=20):\n    file_path = \"best_model.hdf5\"\n    check_point = ModelCheckpoint(file_path, monitor = \"val_loss\", verbose = 1,\n                                  save_best_only = True, mode = \"min\")\n    early_stop = EarlyStopping(monitor = \"val_loss\", mode = \"min\", patience = 3)\n\n    inp = Input(shape = (max_len,))\n    x = Embedding(max_features + 1, embed_size, weights = [embedding_matrix], trainable = False)(inp)\n    x1 = SpatialDropout1D(spatial_dr)(x)\n    \n    # Bidirectional：RNN的双向封装器，对序列进行前向和后向计算（return_sequences:是返回输出序列中的最后一个输出，还是全部序列)\n    x_gru = Bidirectional(CuDNNGRU(units, return_sequences = True))(x1)\n    x_lstm = Bidirectional(CuDNNLSTM(units, return_sequences = True))(x1)\n    \n    # GRU的两个卷积核\n    x_conv1 = Conv1D(conv_size, kernel_size=kernel_size1, padding='valid', kernel_initializer='he_uniform')(x_gru)\n    avg_pool1_gru = GlobalAveragePooling1D()(x_conv1)\n    max_pool1_gru = GlobalMaxPooling1D()(x_conv1)\n    \n    x_conv2 = Conv1D(conv_size, kernel_size=kernel_size2, padding='valid', kernel_initializer='he_uniform')(x_gru)\n    avg_pool2_gru = GlobalAveragePooling1D()(x_conv2)\n    max_pool2_gru = GlobalMaxPooling1D()(x_conv2)\n    \n    # LSTM的两个卷积核\n    x_conv3 = Conv1D(conv_size, kernel_size=kernel_size1, padding='valid', kernel_initializer='he_uniform')(x_lstm)\n    avg_pool1_lstm = GlobalAveragePooling1D()(x_conv3)\n    max_pool1_lstm = GlobalMaxPooling1D()(x_conv3)\n    \n    x_conv4 = Conv1D(conv_size, kernel_size=kernel_size2, padding='valid', kernel_initializer='he_uniform')(x_lstm)\n    avg_pool2_lstm = GlobalAveragePooling1D()(x_conv4)\n    max_pool2_lstm = GlobalMaxPooling1D()(x_conv4)\n    \n    \n    x = concatenate([avg_pool1_gru, max_pool1_gru, avg_pool2_gru, max_pool2_gru,\n                    avg_pool1_lstm, max_pool1_lstm, avg_pool2_lstm, max_pool2_lstm])\n    x = BatchNormalization()(x)\n    x = Dropout(dr)(Dense(dense_units, activation='relu') (x))\n    x = BatchNormalization()(x)\n    x = Dropout(dr)(Dense(int(dense_units / 2), activation='relu') (x))\n    x = Dense(2, activation = \"sigmoid\")(x)\n    model = Model(inputs = inp, outputs = x)\n    model.compile(loss = \"binary_crossentropy\", optimizer = Adam(lr = lr, decay = lr_d), metrics = [\"accuracy\"])\n    model.summary()\n    history = model.fit(X_train, y_ohe, batch_size = 512, epochs = epochs, validation_split=0.1, \n                        verbose = 1, callbacks = [check_point, early_stop])\n    model = load_model(file_path)\n    return model","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"90ac92a570fe3c180252dfcf1f22c5f98a08e615"},"cell_type":"code","source":"%%time\nmodel = build_model(lr = 1e-4, lr_d = 0, units = 64, spatial_dr = 0.5, kernel_size1=4, kernel_size2=3, dense_units=16, dr=0.1, conv_size=16, epochs=5)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"8b206256803deeb5ba630578da1175b3cc01f555"},"cell_type":"code","source":"# pred = model.predict(X_test, batch_size = 1024, verbose = 1)\n# predictions = np.round(np.argmax(pred, axis=1)).astype(int)\n# sub['prediction'] = predictions\n# sub.to_csv(\"submission.csv\", index=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"754211d82f222f50cbb19712aca94984b2191588"},"cell_type":"code","source":"def build_model1(lr=0.0, lr_d=0.0, units=0, spatial_dr=0.0, kernel_size1=3, kernel_size2=2, dense_units=128, dr=0.1, conv_size=32, epochs=20):\n    file_path = \"best_model.hdf5\"\n    check_point = ModelCheckpoint(file_path, monitor = \"val_loss\", verbose = 1,\n                                  save_best_only = True, mode = \"min\")\n    early_stop = EarlyStopping(monitor = \"val_loss\", mode = \"min\", patience = 3)\n\n    inp = Input(shape = (max_len,))\n    x = Embedding(max_features + 1, embed_size, weights = [embedding_matrix], trainable = False)(inp)\n    x1 = SpatialDropout1D(spatial_dr)(x)\n    \n    # GRU建模\n    x_gru = Bidirectional(CuDNNGRU(units, return_sequences = True))(x1)\n    \n    x_conv1 = Conv1D(conv_size, kernel_size=kernel_size1, padding='valid', kernel_initializer='he_uniform')(x_gru)\n    avg_pool1_gru = GlobalAveragePooling1D()(x_conv1)\n    max_pool1_gru = GlobalMaxPooling1D()(x_conv1)\n    \n    x_conv2 = Conv1D(conv_size, kernel_size=kernel_size2, padding='valid', kernel_initializer='he_uniform')(x_gru)\n    avg_pool2_gru = GlobalAveragePooling1D()(x_conv2)\n    max_pool2_gru = GlobalMaxPooling1D()(x_conv2)\n\n    \n    \n    x = concatenate([avg_pool1_gru, max_pool1_gru, avg_pool2_gru, max_pool2_gru])\n    x = BatchNormalization()(x)\n    x = Dropout(dr)(Dense(dense_units, activation='relu') (x))\n    x = BatchNormalization()(x)\n    x = Dropout(dr)(Dense(int(dense_units / 2), activation='relu') (x))\n    x = Dense(2, activation = \"sigmoid\")(x)\n    model = Model(inputs = inp, outputs = x)\n    model.compile(loss = \"binary_crossentropy\", optimizer = Adam(lr = lr, decay = lr_d), metrics = [\"accuracy\"])\n    model.summary()\n    history = model.fit(X_train, y_ohe, batch_size = 512, epochs = epochs, validation_split=0.1, \n                        verbose = 1, callbacks = [check_point, early_stop])\n    model = load_model(file_path)\n    return model","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"552f593475f579d8c2822ee6c2d6b52ed1abc3e5"},"cell_type":"code","source":"#model1 = build_model1(lr = 1e-4, lr_d = 1e-7, units = 128, spatial_dr = 0.3, kernel_size1=4, kernel_size2=3, dense_units=32, dr=0.3, conv_size=32, epochs=5)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"0055ab6717740b4b234ada9aacb0c78074e64b55"},"cell_type":"code","source":"#model1_1 = build_model1(lr = 1e-4, lr_d = 1e-7, units = 128, spatial_dr = 0.3, kernel_size1=4, kernel_size2=3, dense_units=32, dr=0.1, conv_size=32, epochs=5)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"836e8eec34c09dcca660d35f39ddd5ec4801147a"},"cell_type":"code","source":"def build_model2(lr=0.0, lr_d=0.0, units=0, spatial_dr=0.0, kernel_size1=3, kernel_size2=2, dense_units=128, dr=0.1, conv_size=32, epochs=20):\n    file_path = \"best_model.hdf5\"\n    check_point = ModelCheckpoint(file_path, monitor = \"val_loss\", verbose = 1,\n                                  save_best_only = True, mode = \"min\")\n    early_stop = EarlyStopping(monitor = \"val_loss\", mode = \"min\", patience = 3)\n\n    inp = Input(shape = (max_len,))\n    x = Embedding(max_features + 1, embed_size, weights = [embedding_matrix], trainable = False)(inp)\n    x1 = SpatialDropout1D(spatial_dr)(x)\n\n    x_gru = Bidirectional(CuDNNGRU(units * 2, return_sequences = True))(x1)\n    x_gru = Bidirectional(CuDNNGRU(units, return_sequences = True))(x_gru)\n    \n    x_conv1 = Conv1D(conv_size, kernel_size=kernel_size1, padding='valid', kernel_initializer='he_uniform')(x_gru)\n    avg_pool1_gru = GlobalAveragePooling1D()(x_conv1)\n    max_pool1_gru = GlobalMaxPooling1D()(x_conv1)\n    \n    x_conv2 = Conv1D(conv_size, kernel_size=kernel_size2, padding='valid', kernel_initializer='he_uniform')(x_gru)\n    avg_pool2_gru = GlobalAveragePooling1D()(x_conv2)\n    max_pool2_gru = GlobalMaxPooling1D()(x_conv2)\n    \n    x = concatenate([avg_pool1_gru, max_pool1_gru, avg_pool2_gru, max_pool2_gru])\n    x = BatchNormalization()(x)\n    x = Dropout(dr)(Dense(dense_units, activation='relu') (x))\n    x = BatchNormalization()(x)\n    x = Dropout(dr)(Dense(int(dense_units / 2), activation='relu') (x))\n    x = Dense(2, activation = \"sigmoid\")(x)\n    model = Model(inputs = inp, outputs = x)\n    model.compile(loss = \"binary_crossentropy\", optimizer = Adam(lr = lr, decay = lr_d), metrics = [\"accuracy\"])\n    model.summary()\n    history = model.fit(X_train, y_ohe, batch_size = 512, epochs = epochs, validation_split=0.1, \n                        verbose = 1, callbacks = [check_point, early_stop])\n    model = load_model(file_path)\n    return model","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"0284b87509f5b94a6d03db579be74dad0bc8ff73"},"cell_type":"code","source":"#%%time\n#model2 = build_model2(lr = 1e-4, lr_d = 1e-7, units = 256, spatial_dr = 0.3, kernel_size1=4, kernel_size2=3, dense_units=32, dr=0.1, conv_size=32, epochs=5)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ac612d038ef7cf161c4e7c678751d29e4e2c7a82"},"cell_type":"code","source":"#model3 = build_model2(lr = 1e-3, lr_d = 1e-7, units = 256, spatial_dr = 0.3, kernel_size1=4, kernel_size2=3, dense_units=32, dr=0.1, conv_size=16, epochs=5)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"3f6b13c224c17bf70eac09f9bc3846d8332fa8d4"},"cell_type":"code","source":"###%%time\n###model4 = build_model2(lr = 1e-4, lr_d = 1e-7, units = 64, spatial_dr = 0.3, kernel_size1=4, kernel_size2=3, dense_units=32, dr=0.1, conv_size=8, epochs=5)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"582cadcbd9fdd70e4c6da7160d743057fd9a487e"},"cell_type":"code","source":"#model5 = build_model2(lr = 1e-4, lr_d = 1e-7, units = 256, spatial_dr = 0.1, kernel_size1=4, kernel_size2=3, dense_units=32, dr=0.1, conv_size=16, epochs=5)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"04d92de6fc799aca62d3b41c588973d2480b2672"},"cell_type":"markdown","source":"### Model with attention\n\nhttps://github.com/Diyago/ML-DL-scripts/blob/9e161a96580efa9993805ca28f610df72fe36406/DEEP%20LEARNING/LSTM%20RNN/Sentiment%20analysis%20LSTM%20wth%20Bidirectional%20%20%2B%20Custom%20Attention.ipynb"},{"metadata":{"trusted":true,"_uuid":"0571d9edafb014eb480950a8c67fb892a4780240"},"cell_type":"code","source":"class Attention(Layer):\n    def __init__(self, step_dim,\n                 W_regularizer=None, b_regularizer=None,\n                 W_constraint=None, b_constraint=None,\n                 bias=True, **kwargs):\n        \"\"\"\n        Keras Layer that implements an Attention mechanism for temporal data.\n        Supports Masking.\n        Follows the work of Raffel et al. [https://arxiv.org/abs/1512.08756]\n        # Input shape\n            3D tensor with shape: `(samples, steps, features)`.\n        # Output shape\n            2D tensor with shape: `(samples, features)`.\n        :param kwargs:\n        Just put it on top of an RNN Layer (GRU/LSTM/SimpleRNN) with return_sequences=True.\n        The dimensions are inferred based on the output shape of the RNN.\n        Example:\n            model.add(LSTM(64, return_sequences=True))\n            model.add(Attention())\n        \"\"\"\n        self.supports_masking = True\n        #self.init = initializations.get('glorot_uniform')\n        self.init = initializers.get('glorot_uniform')\n\n        self.W_regularizer = regularizers.get(W_regularizer)\n        self.b_regularizer = regularizers.get(b_regularizer)\n\n        self.W_constraint = constraints.get(W_constraint)\n        self.b_constraint = constraints.get(b_constraint)\n\n        self.bias = bias\n        self.step_dim = step_dim\n        self.features_dim = 0\n        super(Attention, self).__init__(**kwargs)\n\n    def build(self, input_shape):\n        assert len(input_shape) == 3\n\n        self.W = self.add_weight((input_shape[-1],),\n                                 initializer=self.init,\n                                 name='{}_W'.format(self.name),\n                                 regularizer=self.W_regularizer,\n                                 constraint=self.W_constraint)\n        self.features_dim = input_shape[-1]\n\n        if self.bias:\n            self.b = self.add_weight((input_shape[1],),\n                                     initializer='zero',\n                                     name='{}_b'.format(self.name),\n                                     regularizer=self.b_regularizer,\n                                     constraint=self.b_constraint)\n        else:\n            self.b = None\n\n        self.built = True\n\n    def compute_mask(self, input, input_mask=None):\n        # do not pass the mask to the next layers\n        return None\n\n    def call(self, x, mask=None):\n        # eij = K.dot(x, self.W) TF backend doesn't support it\n\n        # features_dim = self.W.shape[0]\n        # step_dim = x._keras_shape[1]\n\n        features_dim = self.features_dim\n        step_dim = self.step_dim\n\n        eij = K.reshape(K.dot(K.reshape(x, (-1, features_dim)), K.reshape(self.W, (features_dim, 1))), (-1, step_dim))\n\n        if self.bias:\n            eij += self.b\n\n        eij = K.tanh(eij)\n\n        a = K.exp(eij)\n\n        # apply mask after the exp. will be re-normalized next\n        if mask is not None:\n            # Cast the mask to floatX to avoid float64 upcasting in theano\n            a *= K.cast(mask, K.floatx())\n\n        # in some cases especially in the early stages of training the sum may be almost zero\n        a /= K.cast(K.sum(a, axis=1, keepdims=True) + K.epsilon(), K.floatx())\n\n        a = K.expand_dims(a)\n        weighted_input = x * a\n    #print weigthted_input.shape\n        return K.sum(weighted_input, axis=1)\n\n    def compute_output_shape(self, input_shape):\n        #return input_shape[0], input_shape[-1]\n        return input_shape[0],  self.features_dim","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"143768540a1c3ad0783e49362e9018b7d46a1f07"},"cell_type":"markdown","source":"### Attention"},{"metadata":{"trusted":true,"_uuid":"19030bd2b6277b9e5dbc6933ca0abcf68fc7b3f9"},"cell_type":"code","source":"def build_model3(lr=0.0, lr_d=0.0, units=0, spatial_dr=0.0, dense_units=128, dr=0.1, use_attention=True):\n    inp = Input(shape = (max_len,))\n    x = Embedding(max_features + 1, embed_size, weights = [embedding_matrix], trainable = False)(inp)\n    x1 = SpatialDropout1D(spatial_dr)(x)\n\n    x_gru = Bidirectional(CuDNNGRU(units * 2, return_sequences = True))(x1)\n    if use_attention:\n        x_att = Attention(max_len)(x_gru)\n        x = Dropout(dr)(Dense(dense_units, activation='relu') (x_att))\n    else:\n        x_att = Flatten() (x_gru)\n        x = Dropout(dr)(Dense(dense_units, activation='relu') (x_att))\n\n    x = BatchNormalization()(x)\n    #x = Dropout(dr)(Dense(int(dense_units / 2), activation='relu') (x))\n    x = Dense(2, activation = \"sigmoid\")(x)\n    model = Model(inputs = inp, outputs = x)\n    model.compile(loss = \"binary_crossentropy\", optimizer = Adam(lr = lr, decay = lr_d), metrics = [\"accuracy\"])\n    #model.summary()\n    #history = model.fit(X_train, y_ohe, batch_size = 512, epochs = epochs, validation_split=0.1, \n    #                    verbose = 1, callbacks = [check_point, early_stop])\n    #model = load_model(file_path)\n    return model","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"1e7072209b383437e44560d062c1614418a1607a"},"cell_type":"code","source":"# %%time\n# file_path = \"best_model.hdf5\"\n# check_point = ModelCheckpoint(file_path, monitor = \"val_loss\", verbose = 1,\n#                               save_best_only = True, mode = \"min\")\n# early_stop = EarlyStopping(monitor = \"val_loss\", mode = \"min\", patience = 3)\n# model6 = build_model3(lr = 1e-3, lr_d = 1e-7, units = 64, spatial_dr = 0.3, dense_units=16, dr=0.1, use_attention=True)\n# history = model6.fit(X_train, y_ohe, batch_size = 512, epochs = 10, validation_split=0.1, \n#                     verbose = 1, callbacks = [check_point, early_stop])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"4c0cab2368cf13a33e56a92eec20ac01254c3529"},"cell_type":"code","source":"#%%time\nfile_path = \"best_model.hdf5\"\ncheck_point = ModelCheckpoint(file_path, monitor = \"val_loss\", verbose = 1,\n                              save_best_only = True, mode = \"min\")\nearly_stop = EarlyStopping(monitor = \"val_loss\", mode = \"min\", patience = 3)\nmodel3 = build_model3(lr = 1e-3, lr_d = 1e-7, units = 64, spatial_dr = 0.3, dense_units=16, dr=0.1, use_attention=True)\nhistory = model3.fit(X_train, y_ohe, batch_size = 512, epochs = 5, validation_split=0.1, \n                    verbose = 1, callbacks = [check_point, early_stop])","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"bbc608b1a70e5bc21df515136b6225da29e797a9"},"cell_type":"markdown","source":"### One branch"},{"metadata":{"trusted":true,"_uuid":"fecf3405fccd307de507cea500df6b559e70cc95"},"cell_type":"code","source":"def build_model4(lr=0.0, lr_d=0.0, units=0, spatial_dr=0.0, kernel_size1=3, kernel_size2=2, dense_units=128, dr=0.1, conv_size=32, epochs=20):\n    file_path = \"best_model.hdf5\"\n    check_point = ModelCheckpoint(file_path, monitor = \"val_loss\", verbose = 1,\n                                  save_best_only = True, mode = \"min\")\n    early_stop = EarlyStopping(monitor = \"val_loss\", mode = \"min\", patience = 3)\n\n    inp = Input(shape = (max_len,))\n    x = Embedding(max_features + 1, embed_size, weights = [embedding_matrix], trainable = False)(inp)\n    x1 = SpatialDropout1D(spatial_dr)(x)\n\n    x_gru = Bidirectional(CuDNNGRU(units, return_sequences = True))(x1)\n    \n    x_conv1 = Conv1D(conv_size, kernel_size=kernel_size1, padding='valid', kernel_initializer='he_uniform')(x_gru)\n    avg_pool1_gru = GlobalAveragePooling1D()(x_conv1)\n    max_pool1_gru = GlobalMaxPooling1D()(x_conv1)\n       \n    x = concatenate([avg_pool1_gru, max_pool1_gru])\n    x = BatchNormalization()(x)\n    x = Dropout(dr)(Dense(dense_units, activation='relu') (x))\n    x = BatchNormalization()(x)\n    #x = Dropout(dr)(Dense(int(dense_units / 2), activation='relu') (x))\n    x = Dense(2, activation = \"sigmoid\")(x)\n    model = Model(inputs = inp, outputs = x)\n    model.compile(loss = \"binary_crossentropy\", optimizer = Adam(lr = lr, decay = lr_d), metrics = [\"accuracy\"])\n    model.summary()\n    history = model.fit(X_train, y_ohe, batch_size = 512, epochs = epochs, validation_split=0.1, \n                        verbose = 1, callbacks = [check_point, early_stop])\n    model = load_model(file_path)\n    return model","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"29fcf5b25f20763deb75612eac29d58add377fee"},"cell_type":"code","source":"###%%time\n###model4 = build_model4(lr = 1e-4, lr_d = 1e-7, units = 64, spatial_dr = 0.3, kernel_size1=3, dense_units=32, dr=0.1, conv_size=8, epochs=5)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"1b21c1b068660aaca49e9f1d1c8259bd5c97ad56"},"cell_type":"code","source":"#model8 = build_model4(lr = 1e-4, lr_d = 1e-7, units = 128, spatial_dr = 0.3, kernel_size1=4, dense_units=32, dr=0.1, conv_size=8, epochs=5)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"f12bb4ca1ee7bca3caa41426590330365e057d34"},"cell_type":"markdown","source":"### Masking"},{"metadata":{"trusted":true,"_uuid":"dcae5f568eac0945c349eb2253df8010d03faa97"},"cell_type":"code","source":"def build_model5(lr=0.0, lr_d=0.0, units=0, spatial_dr=0.0, kernel_size1=3, kernel_size2=2, dense_units=128, dr=0.1, conv_size=32, epochs=20):\n    file_path = \"best_model.hdf5\"\n    check_point = ModelCheckpoint(file_path, monitor = \"val_loss\", verbose = 1,\n                                  save_best_only = True, mode = \"min\")\n    early_stop = EarlyStopping(monitor = \"val_loss\", mode = \"min\", patience = 3)\n\n    inp = Input(shape = (max_len,))\n    x = Embedding(max_features + 1, embed_size, weights = [embedding_matrix], trainable = False)(inp)\n    x1 = SpatialDropout1D(spatial_dr)(x)\n    x_m = Masking()(x1) # 使用给定的值对输入的序列信号进行“屏蔽”，用以定位需要跳过的时间步（将缺省的时间步的值赋值为0）\n    x_gru = LSTM(units)(x_m)\n\n    x = BatchNormalization()(x_gru)\n    x = Dropout(dr)(Dense(dense_units, activation='relu') (x))\n    x = BatchNormalization()(x)\n    #x = Dropout(dr)(Dense(int(dense_units / 2), activation='relu') (x))\n    x = Dense(2, activation = \"sigmoid\")(x)\n    model = Model(inputs = inp, outputs = x)\n    model.compile(loss = \"binary_crossentropy\", optimizer = Adam(lr = lr, decay = lr_d), metrics = [\"accuracy\"])\n    model.summary()\n    history = model.fit(X_train, y_ohe, batch_size = 512, epochs = epochs, validation_split=0.1, \n                        verbose = 1, callbacks = [check_point, early_stop])\n    model = load_model(file_path)\n    return model","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"3b98e6c57a2f16b29b8ae7265919eb309e36e770"},"cell_type":"code","source":"###model5 = build_model5(lr = 1e-4, lr_d = 1e-7, units = 128, spatial_dr = 0.3, kernel_size1=4, dense_units=32, dr=0.1, conv_size=8, epochs=10)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"771278144d9fe02f509ee4e08b7a97d3e69b8283"},"cell_type":"markdown","source":"print('Start predicting...\\n')\npred1 = model.predict(X_test, batch_size = 1024, verbose = 1)\npred = pred1\npred2 = model4.predict(X_test, batch_size = 1024, verbose = 1)\npred += pred2\npred3 = model5.predict(X_test, batch_size = 1024, verbose = 1)\npred += pred3\npred4 = model3.predict(X_test, batch_size = 1024, verbose = 1)\npred += pred4\n\npred = pred / 4"},{"metadata":{"trusted":true,"_uuid":"a932d7db67949d7ffcc6ba5be64ab232e29794cb"},"cell_type":"code","source":"pred = model3.predict(X_test, batch_size = 1024, verbose = 1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"2c04e298f5c96a9561450874a0238f408984c39e"},"cell_type":"code","source":"#pred = model9.predict(X_test, batch_size = 1024, verbose = 1)\n\npredictions = np.round(np.argmax(pred, axis=1)).astype(int)\nsub['prediction'] = predictions\nsub.to_csv(\"submission.csv\", index=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"0ff7c50a702a776884cee0b6cb49daa6159cb0b9"},"cell_type":"code","source":"sub.to_csv(\"submission_tag.csv\", index=False)\nprint('Saved submission.')","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}