{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nimport os\nprint(os.listdir(\"../input\"))\n\n# Any results you write to the current directory are saved as output.\nimport numpy as np\nimport pandas as pd\n\nfrom tqdm import tqdm\n\n\nfrom sklearn.model_selection import train_test_split\nfrom sklearn import metrics\n\nfrom keras.preprocessing.text import Tokenizer\nfrom keras.preprocessing.sequence import pad_sequences\n\nfrom keras import backend as K\nfrom keras.engine.topology import Layer\nfrom keras.optimizers import Adam\nfrom keras.models import Sequential\nfrom keras.layers import LSTM, Bidirectional, GlobalMaxPool1D, Dropout,Activation,Conv1D,CuDNNLSTM,CuDNNGRU\nfrom keras.layers import MaxPooling1D,BatchNormalization,Conv2D,Flatten,GlobalAveragePooling1D,GlobalMaxPooling1D\nfrom keras.layers import Dense, Embedding, Input,SpatialDropout1D,concatenate\nfrom keras import initializers, regularizers, constraints, optimizers, layers","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"train_df = pd.read_csv(\"../input/train.csv\")\ntest_df = pd.read_csv(\"../input/test.csv\")\nprint(\"Train shape : \",train_df.shape)\nprint(\"Test shape : \",test_df.shape)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"3d2fbd2688f8455fe558e9398b7b6a8dca0fba90"},"cell_type":"code","source":"train_df, val_df = train_test_split(train_df, test_size=0.1, random_state=2018)\nembed_size = 300 \nmax_features = 50000 \nmaxlen = 100 ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"05b0f0a1544b5f5c7ef43951236e9ff5f23e19a8"},"cell_type":"code","source":"train_X = train_df[\"question_text\"].fillna(\"_na_\").values\nval_X = val_df[\"question_text\"].fillna(\"_na_\").values\ntest_X = test_df[\"question_text\"].fillna(\"_na_\").values","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"be434771a64ee2ca4bb1568fb57dab5c8d809e01"},"cell_type":"code","source":"## Tokenize the sentences\ntokenizer = Tokenizer(num_words=max_features)\ntokenizer.fit_on_texts(list(train_X))\ntrain_X = tokenizer.texts_to_sequences(train_X)\nval_X = tokenizer.texts_to_sequences(val_X)\ntest_X = tokenizer.texts_to_sequences(test_X)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"1e893f22dfd0a7740a6692fb7fe6efda6b31b427"},"cell_type":"code","source":"## Pad the sentences \ntrain_X = pad_sequences(train_X, maxlen=maxlen)\nval_X = pad_sequences(val_X, maxlen=maxlen)\ntest_X = pad_sequences(test_X, maxlen=maxlen)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"57f1e81f92cb5a6f84e92bf3dad2cf57adc5b1ad"},"cell_type":"code","source":"## Get the target values\ntrain_y = train_df['target'].values\nval_y = val_df['target'].values","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"18d8a5d8e2581e3ab9f37da88bfa7123ccb70bdc"},"cell_type":"code","source":"embeddings_index = {}\nf = open('../input/embeddings/glove.840B.300d/glove.840B.300d.txt')\nfor line in tqdm(f):\n    values = line.split(\" \")\n    word = values[0]\n    coefs = np.asarray(values[1:], dtype='float32')\n    embeddings_index[word] = coefs\nf.close()\n\nprint('Found %s word vectors.' % len(embeddings_index))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"0274882b35d499886f79988e66310310ac9c7427"},"cell_type":"code","source":"all_embs = np.stack(embeddings_index.values())\nemb_mean,emb_std = all_embs.mean(), all_embs.std()\nembed_size = all_embs.shape[1]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a64b9c8698fc17f2bb44ba2e454d3eb228c2441f"},"cell_type":"code","source":"word_index = tokenizer.word_index\nnb_words = min(max_features, len(word_index))\nembedding_matrix = np.random.normal(emb_mean, emb_std, (nb_words, embed_size))\nfor word, i in word_index.items():\n    if i >= max_features: continue\n    embedding_vector = embeddings_index.get(word)\n    if embedding_vector is not None: embedding_matrix[i] = embedding_vector","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"7704227a1056dcfe50decea5850beeea1cd9014b"},"cell_type":"markdown","source":"Attention Layer: https://www.kaggle.com/suicaokhoailang/lstm-attention-baseline-0-652-lb"},{"metadata":{"trusted":true,"_uuid":"5af47778f40f3b67cd2be90cac8aadbabf9eaf8e"},"cell_type":"code","source":"class Attention(Layer):\n    def __init__(self, step_dim,\n                 W_regularizer=None, b_regularizer=None,\n                 W_constraint=None, b_constraint=None,\n                 bias=True, **kwargs):\n        self.supports_masking = True\n        self.init = initializers.get('glorot_uniform')\n\n        self.W_regularizer = regularizers.get(W_regularizer)\n        self.b_regularizer = regularizers.get(b_regularizer)\n\n        self.W_constraint = constraints.get(W_constraint)\n        self.b_constraint = constraints.get(b_constraint)\n\n        self.bias = bias\n        self.step_dim = step_dim\n        self.features_dim = 0\n        super(Attention, self).__init__(**kwargs)\n\n    def build(self, input_shape):\n        assert len(input_shape) == 3\n\n        self.W = self.add_weight((input_shape[-1],),\n                                 initializer=self.init,\n                                 name='{}_W'.format(self.name),\n                                 regularizer=self.W_regularizer,\n                                 constraint=self.W_constraint)\n        self.features_dim = input_shape[-1]\n\n        if self.bias:\n            self.b = self.add_weight((input_shape[1],),\n                                     initializer='zero',\n                                     name='{}_b'.format(self.name),\n                                     regularizer=self.b_regularizer,\n                                     constraint=self.b_constraint)\n        else:\n            self.b = None\n\n        self.built = True\n\n    def compute_mask(self, input, input_mask=None):\n        return None\n\n    def call(self, x, mask=None):\n        features_dim = self.features_dim\n        step_dim = self.step_dim\n\n        eij = K.reshape(K.dot(K.reshape(x, (-1, features_dim)),\n                        K.reshape(self.W, (features_dim, 1))), (-1, step_dim))\n\n        if self.bias:\n            eij += self.b\n\n        eij = K.tanh(eij)\n\n        a = K.exp(eij)\n\n        if mask is not None:\n            a *= K.cast(mask, K.floatx())\n\n        a /= K.cast(K.sum(a, axis=1, keepdims=True) + K.epsilon(), K.floatx())\n\n        a = K.expand_dims(a)\n        weighted_input = x * a\n        return K.sum(weighted_input, axis=1)\n\n    def compute_output_shape(self, input_shape):\n        return input_shape[0],  self.features_dim","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"df08195d8933b050d446105f7ec3de3d0f060900"},"cell_type":"code","source":"model=Sequential()\nmodel.add(Embedding(max_features, embed_size, weights=[embedding_matrix],input_length=maxlen,trainable = False))\nmodel.add(Bidirectional(CuDNNLSTM(128, return_sequences=True)))\nmodel.add(Bidirectional(CuDNNLSTM(64, return_sequences=True)))\nmodel.add(Attention(maxlen))\n#model.add(GlobalMaxPool1D())\nmodel.add(Dense(64, activation=\"relu\"))\nmodel.add(Dropout(0.2))\nmodel.add(Dense(1, activation=\"sigmoid\"))\nmodel.compile(loss='binary_crossentropy', optimizer=Adam(lr=1e-3), metrics=['accuracy'])\nprint(model.summary())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"aee6172f162419ed066de0c6f9487b1d8477ad75"},"cell_type":"code","source":"model.fit(train_X, train_y, batch_size=512, epochs=3, validation_data=(val_X, val_y))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"7c24731a35251807160c823d3285fc4a348c8393"},"cell_type":"code","source":"model.evaluate(x=val_X, y=val_y, batch_size=1024, verbose=1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d86a831e8d520dbf9c849c2719d3daed92310acb"},"cell_type":"code","source":"test_y_prediction = model.predict(test_X, batch_size = 1024, verbose = 1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"73696de5671a1910e4c65de564e95362f775f56c"},"cell_type":"code","source":"pred_val_y = model.predict([val_X], batch_size=1024, verbose=1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"fac667b1e184505f62e032fee4dd2977767eb767"},"cell_type":"code","source":"for thresh in np.arange(0.1, 0.501, 0.01):\n    thresh = np.round(thresh, 2)\n    print(\"F1 score at threshold {0} is {1}\".format(thresh, metrics.f1_score(val_y, (pred_val_y>thresh).astype(int))))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b054eb871d3bdc5db1680f087d4176921e097540"},"cell_type":"code","source":"pred_test_y = (test_y_prediction>0.40).astype(int)\nout_df = pd.DataFrame({\"qid\":test_df[\"qid\"].values})\nout_df['prediction'] = pred_test_y\nout_df.to_csv(\"submission.csv\", index=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"677c8617f7a51620a1b489e7e4a6c2641b4c65bb"},"cell_type":"markdown","source":"**Refrences:**\nhttps://www.kaggle.com/sudalairajkumar/a-look-at-different-embeddings \nhttps://www.kaggle.com/suicaokhoailang/lstm-attention-baseline-0-652-lb"},{"metadata":{"trusted":true,"_uuid":"efa517cab43359ce661f9c19a33d48271d38f9f8"},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}