{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport os\nimport time\nimport numpy as np \nimport pandas as pd \nfrom tqdm import tqdm\nimport math\nfrom sklearn.model_selection import train_test_split\nfrom sklearn import metrics\n\nfrom keras import backend as K\nfrom keras.preprocessing.text import Tokenizer\nfrom keras.preprocessing.sequence import pad_sequences\nfrom keras.layers import Dense, Input, LSTM, Embedding, Dropout, Activation, CuDNNGRU, Conv1D, concatenate\nfrom keras.layers import ConvRNN2D, SpatialDropout1D, Reshape, MaxPool2D, Concatenate, Flatten, Conv2D\nfrom keras.layers import Bidirectional, GlobalMaxPool1D\nfrom keras.models import Model\nfrom keras import initializers, regularizers, constraints, optimizers, layers\nimport matplotlib.pyplot as plt\nfrom tensorflow.python.client import device_lib\n%matplotlib inline\n\nfrom keras.models import Model\nfrom keras import initializers, regularizers, constraints, optimizers, layers\nfrom keras.engine.topology import Layer\n\n#GPU configs\nos.environ['CUDA_VISIBLE_DEVICES'] = \"0\"\n\ngpu_options = K.tf.GPUOptions(per_process_gpu_memory_fraction = 1)\nconfig = K.tf.ConfigProto(gpu_options = gpu_options, allow_soft_placement = True)\nK.set_session(K.tf.Session(config = config))\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nimport os\nprint(os.listdir(\"../input\"))\n\n# Any results you write to the current directory are saved as output.\n\n#Source: https://www.kaggle.com/sudalairajkumar/a-look-at-different-embeddings\nprint([dev.name for dev in device_lib.list_local_devices()])","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"7a4abf2c40832f23048fc6371f048ede1ffecfab"},"cell_type":"markdown","source":"### Load the Data"},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"%%time\nbase_path = \"../input/\"\ntrain_df = pd.read_csv(base_path+\"train.csv\")\ntest_df = pd.read_csv(base_path+\"test.csv\")\nprint(\"Train shape : \",train_df.shape)\nprint(\"Test shape : \",test_df.shape)\n\n\n## split to train and val\ntrain_df, val_df = train_test_split(train_df, test_size=0.1, random_state=2018)\n\n## some config values \nembed_size = 300 # how big is each word vector\nmax_features = 50000 # how many unique words to use (i.e num rows in embedding vector)\nmaxlen = 100 # max number of words in a question to use\n\n## fill up the missing values\ntrain_X = train_df[\"question_text\"].fillna(\"_na_\").values\nval_X = val_df[\"question_text\"].fillna(\"_na_\").values\ntest_X = test_df[\"question_text\"].fillna(\"_na_\").values\n\n## Tokenize the sentences\ntokenizer = Tokenizer(num_words=max_features)\ntokenizer.fit_on_texts(list(train_X))\ntrain_X = tokenizer.texts_to_sequences(train_X)\nval_X = tokenizer.texts_to_sequences(val_X)\ntest_X = tokenizer.texts_to_sequences(test_X)\n\n## Pad the sentences \ntrain_X = pad_sequences(train_X, maxlen=maxlen)\nval_X = pad_sequences(val_X, maxlen=maxlen)\ntest_X = pad_sequences(test_X, maxlen=maxlen)\n\n## Get the target values\ntrain_y = train_df['target'].values\nval_y = val_df['target'].values\n\nprint(\"Train: \", train_X.shape, train_y.shape)\nprint(\"Validation: \", val_X.shape, val_y.shape)\nprint(\"Test :\", test_X.shape)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"446b56e44511a11997524293bad56e7d3256e351"},"cell_type":"markdown","source":"### Load the Embeddings"},{"metadata":{"trusted":true,"_uuid":"4852bf1808ffe57a0077a9621d6d709007c2485f"},"cell_type":"code","source":"%%time\ndef get_coefs(word,*arr): \n    return word, np.asarray(arr, dtype='float32')\n\nbase_path = \"../input/embeddings/\"\nfiles = {\"glove\": \"glove.840B.300d/glove.840B.300d.txt\",\n         \"wiki_news\": \"wiki-news-300d-1M/wiki-news-300d-1M.vec\",\n         \"paragram\": \"paragram_300_sl999/paragram_300_sl999.txt\"}\n\nembedding_matrices = {}\n\nfor emb in files:\n    if emb==\"glove\":\n        emb_index = dict(get_coefs(*o.split(\" \")) for o in open(base_path+files[emb]))\n    elif emb==\"wiki_news\":\n        emb_index = dict(get_coefs(*o.split(\" \")) for o in open(base_path+files[emb]) \\\n                         if len(o)>100)\n    elif emb==\"paragram\":\n        emb_index = dict(get_coefs(*o.split(\" \")) for o in open(base_path+files[emb],\n                                                             encoding=\"utf8\", \n                                                             errors='ignore') \\\n                         if len(o)>100)\n    all_embs = np.stack(emb_index.values())\n    emb_mean, emb_std = all_embs.mean(), all_embs.std()\n    embed_size = all_embs.shape[1]\n    \n    word_index = tokenizer.word_index\n    nb_words = min(max_features, len(word_index))\n    embedding_matrix = np.random.normal(emb_mean, emb_std, (nb_words, embed_size))\n    for word, i in word_index.items():\n        if i >= max_features: continue\n        embedding_vector = emb_index.get(word)\n        if embedding_vector is not None: embedding_matrix[i] = embedding_vector\n            \n    embedding_matrices[emb] = embedding_matrix","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d80e0886107273ad7de1bcef2eaee7243e8b23cb"},"cell_type":"markdown","source":"### CuDNNGRU network"},{"metadata":{"trusted":true,"scrolled":false,"_uuid":"a7ac0038f40d624df25358d18ad57b4362fd57ad"},"cell_type":"code","source":"%%time\ninp = Input(shape=(maxlen,))\n\ninp_glove = Embedding(max_features, embed_size, weights=[embedding_matrices[\"glove\"]])(inp)\ninp_glove = Bidirectional(CuDNNGRU(128, return_sequences=True))(inp_glove)\ninp_glove = GlobalMaxPool1D()(inp_glove)\n\ninp_wiki = Embedding(max_features, embed_size, weights=[embedding_matrices[\"wiki_news\"]])(inp)\ninp_wiki = Bidirectional(CuDNNGRU(128, return_sequences=True))(inp_wiki)\ninp_wiki = GlobalMaxPool1D()(inp_wiki)\n\ninp_paragram = Embedding(max_features, embed_size, weights=[embedding_matrices[\"paragram\"]])(inp)\ninp_paragram = Bidirectional(CuDNNGRU(128, return_sequences=True))(inp_paragram)\ninp_paragram = GlobalMaxPool1D()(inp_paragram)\n\nmerged  = concatenate([inp_glove, inp_wiki, inp_paragram])\nx = Dense(32, activation=\"relu\")(merged)\nx = Dropout(0.1)(x)\nx = Dense(1, activation=\"sigmoid\")(x)\nmodel = Model(inputs=inp, outputs=x)\nmodel.compile(loss='binary_crossentropy', optimizer='adam', metrics=['accuracy'])\nprint(model.summary())\n\n#train\nmodel.fit(train_X, train_y, batch_size=512, epochs=2, validation_data=(val_X, val_y))\n\n#model performance \npred_val_y = model.predict([val_X], batch_size=1024, verbose=1)\nf1_scores = []\nthreshs = np.arange(0.1, 0.9, 0.01)\nfor thresh in threshs:\n    thresh = np.round(thresh, 2)\n    f1_score = metrics.f1_score(val_y, (pred_val_y>thresh).astype(int))\n    f1_scores.append(f1_score)\n    \nplt.plot(threshs, f1_scores)\nmax_fscore = np.round(f1_scores[np.argmax(f1_scores)], 3)\nmax_thresh = np.round(threshs[np.argmax(f1_scores)], 3)\nplt.title(\"F scores at different values of thresholds | Max: {} | Thresh {}\".format(max_fscore, max_thresh))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"scrolled":true,"_uuid":"dc104351d7b66d267483f9dc73581360d6871b6c"},"cell_type":"code","source":"thresh = max_thresh\n\n#predictions on the validation set\npred_val_y_1 = model.predict([val_X], batch_size=1024, verbose=1)\n#pred_val_y_1 = (pred_val_y_1>=thresh).astype(int)\n\n#predictions on the test set\npred_test_y_1 = model.predict([test_X], batch_size=1024, verbose=1)\n#pred_test_y_1 = (pred_test_y_1>=thresh).astype(int)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"08c6e8c7812b5645e6224789844f8262e8663e69"},"cell_type":"markdown","source":"### CNN architecture"},{"metadata":{"trusted":true,"scrolled":false,"_uuid":"9d1ca27e57a0422ecabbdf3c56826a23e08f81c2"},"cell_type":"code","source":"%%time\nfilter_sizes = [1, 2, 3, 5]\nnum_filt = 36\n\ninp = Input(shape=(maxlen,))\n\nembeddings = [Embedding(max_features, embed_size, weights=[embedding_matrices[emb]])(inp) \\\n              for emb in embedding_matrices]\n\nembed_maxpools = []\nfor embed in embeddings:\n    embed = SpatialDropout1D(0.1)(embed)\n    embed = Reshape((maxlen, embed_size, 1))(embed)\n    \n    maxpools = []\n    for i in range(len(filter_sizes)):\n        conv = Conv2D(num_filt, kernel_size=(filter_sizes[i], embed_size),\n                  kernel_initializer=\"he_normal\", activation=\"elu\")(embed)\n        maxpools.append(MaxPool2D(pool_size=(maxlen - filter_sizes[i] + 1, 1))(conv))\n    merged = Concatenate(axis=1)(maxpools)\n    merged = Flatten()(merged)\n    merged = Dropout(0.2)(merged)\n    \n    embed_maxpools.append(merged)\n\nembed_maxpools = Concatenate(axis=1)(embed_maxpools)\n   \nx = Dense(32, activation=\"relu\")(embed_maxpools)\nx = Dropout(0.1)(x)\nx = Dense(1, activation=\"sigmoid\")(x)\nmodel = Model(inputs=inp, outputs=x)\nmodel.compile(loss='binary_crossentropy', optimizer='adam', metrics=['accuracy'])\nprint(model.summary())\n\n#train\nmodel.fit(train_X, train_y, batch_size=256, epochs=2, validation_data=(val_X, val_y))\n\n#model performance \npred_val_y = model.predict([val_X], batch_size=1024, verbose=1)\nf1_scores = []\nthreshs = np.arange(0.1, 0.9, 0.01)\nfor thresh in threshs:\n    thresh = np.round(thresh, 2)\n    f1_score = metrics.f1_score(val_y, (pred_val_y>thresh).astype(int))\n    f1_scores.append(f1_score)\n    \nplt.plot(threshs, f1_scores)\nmax_fscore = np.round(f1_scores[np.argmax(f1_scores)], 3)\nmax_thresh = np.round(threshs[np.argmax(f1_scores)], 3)\nplt.title(\"F scores at different values of thresholds | Max: {} | Thresh {}\".format(max_fscore, max_thresh))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"2e0ce1aa4af80857d5ae034a54315b15df976c7c"},"cell_type":"code","source":"thresh = max_thresh\n\n#predictions on the validation set\npred_val_y_2 = model.predict([val_X], batch_size=1024, verbose=1)\n#pred_val_y_2 = (pred_val_y_2>=thresh).astype(int)\n\n#predictions on the test set\npred_test_y_2 = model.predict([test_X], batch_size=1024, verbose=1)\n#pred_test_y_2 = (pred_test_y_2>=thresh).astype(int)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"ab67e4a3920b6eaebcf640d0027da685164df37c"},"cell_type":"markdown","source":"### CuDNNGRU architecture with Attention Layer"},{"metadata":{"trusted":true,"_uuid":"02c823f1a7a1dbb66c801bb74df33d3989e0dce1"},"cell_type":"code","source":"class Attention(Layer):\n    def __init__(self, step_dim,\n                 W_regularizer=None, b_regularizer=None,\n                 W_constraint=None, b_constraint=None,\n                 bias=True, **kwargs):\n        self.supports_masking = True\n        self.init = initializers.get('glorot_uniform')\n\n        self.W_regularizer = regularizers.get(W_regularizer)\n        self.b_regularizer = regularizers.get(b_regularizer)\n\n        self.W_constraint = constraints.get(W_constraint)\n        self.b_constraint = constraints.get(b_constraint)\n\n        self.bias = bias\n        self.step_dim = step_dim\n        self.features_dim = 0\n        super(Attention, self).__init__(**kwargs)\n\n    def build(self, input_shape):\n        assert len(input_shape) == 3\n\n        self.W = self.add_weight((input_shape[-1],),\n                                 initializer=self.init,\n                                 name='{}_W'.format(self.name),\n                                 regularizer=self.W_regularizer,\n                                 constraint=self.W_constraint)\n        self.features_dim = input_shape[-1]\n\n        if self.bias:\n            self.b = self.add_weight((input_shape[1],),\n                                     initializer='zero',\n                                     name='{}_b'.format(self.name),\n                                     regularizer=self.b_regularizer,\n                                     constraint=self.b_constraint)\n        else:\n            self.b = None\n\n        self.built = True\n\n    def compute_mask(self, input, input_mask=None):\n        return None\n\n    def call(self, x, mask=None):\n        features_dim = self.features_dim\n        step_dim = self.step_dim\n\n        eij = K.reshape(K.dot(K.reshape(x, (-1, features_dim)),\n                        K.reshape(self.W, (features_dim, 1))), (-1, step_dim))\n\n        if self.bias:\n            eij += self.b\n\n        eij = K.tanh(eij)\n\n        a = K.exp(eij)\n\n        if mask is not None:\n            a *= K.cast(mask, K.floatx())\n\n        a /= K.cast(K.sum(a, axis=1, keepdims=True) + K.epsilon(), K.floatx())\n\n        a = K.expand_dims(a)\n        weighted_input = x * a\n        return K.sum(weighted_input, axis=1)\n\n    def compute_output_shape(self, input_shape):\n        return input_shape[0],  self.features_dim","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b0c6d668ec07887091b5539c01fd39018269555e"},"cell_type":"code","source":"%%time\ninp = Input(shape=(maxlen,))\n\ninp_glove = Embedding(max_features, embed_size, weights=[embedding_matrices[\"glove\"]])(inp)\ninp_glove = Bidirectional(CuDNNGRU(128, return_sequences=True))(inp_glove)\ninp_glove = Attention(maxlen)(inp_glove)\n\ninp_wiki = Embedding(max_features, embed_size, weights=[embedding_matrices[\"wiki_news\"]])(inp)\ninp_wiki = Bidirectional(CuDNNGRU(128, return_sequences=True))(inp_wiki)\ninp_wiki = Attention(maxlen)(inp_wiki)\n\ninp_paragram = Embedding(max_features, embed_size, weights=[embedding_matrices[\"paragram\"]])(inp)\ninp_paragram = Bidirectional(CuDNNGRU(128, return_sequences=True))(inp_paragram)\ninp_paragram = Attention(maxlen)(inp_paragram)\n\nmerged  = concatenate([inp_glove, inp_wiki, inp_paragram])\nx = Dense(32, activation=\"relu\")(merged)\nx = Dropout(0.1)(x)\nx = Dense(16, activation=\"relu\")(x)\nx = Dropout(0.1)(x)\nx = Dense(1, activation=\"sigmoid\")(x)\nmodel = Model(inputs=inp, outputs=x)\nmodel.compile(loss='binary_crossentropy', optimizer='adam', metrics=['accuracy'])\nprint(model.summary())\n\n#train\nmodel.fit(train_X, train_y, batch_size=256, epochs=2, validation_data=(val_X, val_y))\n\n#model performance \npred_val_y = model.predict([val_X], batch_size=1024, verbose=1)\nf1_scores = []\nthreshs = np.arange(0.1, 0.9, 0.01)\nfor thresh in threshs:\n    thresh = np.round(thresh, 2)\n    f1_score = metrics.f1_score(val_y, (pred_val_y>thresh).astype(int))\n    f1_scores.append(f1_score)\n    \nplt.plot(threshs, f1_scores)\nmax_fscore = np.round(f1_scores[np.argmax(f1_scores)], 3)\nmax_thresh = np.round(threshs[np.argmax(f1_scores)], 3)\nplt.title(\"F scores at different values of thresholds | Max: {} | Thresh {}\".format(max_fscore, max_thresh))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f9f0b8c99ba8d5b764ff8aef226313dea4c209ac"},"cell_type":"code","source":"thresh = max_thresh\n\n#predictions on the validation set\npred_val_y_3 = model.predict([val_X], batch_size=1024, verbose=1)\n#pred_val_y_3 = (pred_val_y_3>=thresh).astype(int)\n\n#predictions on the test set\npred_test_y_3 = model.predict([test_X], batch_size=1024, verbose=1)\n#pred_test_y_3 = (pred_test_y_3>=thresh).astype(int)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"67df1e97d4b39d6c9dc8ac8a91858d12bcc93409"},"cell_type":"markdown","source":"#### Ensemble Method on the Predictions"},{"metadata":{"trusted":true,"_uuid":"53bc69953ec0764000464f1653849706f2e62cf9"},"cell_type":"code","source":"pred_val_y = 0.5*pred_val_y_1 + 0.3*pred_val_y_2 + 0.2*pred_val_y_3\nf1_scores = []\nthreshs = np.arange(0.1, 0.9, 0.01)\nfor thresh in threshs:\n    thresh = np.round(thresh, 2)\n    f1_score = metrics.f1_score(val_y, (pred_val_y>thresh).astype(int))\n    f1_scores.append(f1_score)\n    \nplt.plot(threshs, f1_scores)\nmax_fscore = np.round(f1_scores[np.argmax(f1_scores)], 3)\nmax_thresh = np.round(threshs[np.argmax(f1_scores)], 3)\nplt.title(\"F scores at different values of thresholds | Max: {} | Thresh {}\".format(max_fscore, max_thresh))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"26b3300ffbb17b511c8497c0826216a163eb2e72"},"cell_type":"code","source":"pred_test_y = 0.5*pred_test_y_1 + 0.3*pred_test_y_2 + 0.2*pred_test_y_3\npred_test_y = (pred_test_y>=max_thresh).astype(int)\nout_df = pd.DataFrame({\"qid\":test_df[\"qid\"].values})\nout_df['prediction'] = pred_test_y\nout_df.to_csv(\"submission.csv\", index=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"5b9814025d485cf74ff17c9312654e45922bacda"},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}