{"cells":[{"metadata":{"_uuid":"cc406dac1c0854793fe8672741f97f773f829909"},"cell_type":"markdown","source":"* Based on SRK's kernel: https://www.kaggle.com/sudalairajkumar/a-look-at-different-embeddings\n* Concatenate embeddings instead of blending predictions. Added SpatialDropout1D.\n* Added word2vec from https://www.kaggle.com/strideradu/word2vec-and-gensim-go-go-go.\n* Combined Vladimir Demidov's 2DCNN textClassifier: https://www.kaggle.com/yekenot/2dcnn-textclassifier\n* Modified the code to choose best threshold"},{"metadata":{"trusted":true,"_uuid":"144fe5891359c058bba89c6bdef4054d3f324f34"},"cell_type":"code","source":"## Settings:\n    \n# some config values \nembed_size = 300 # how big is each word vector\nmax_features = 50000 # how many unique words to use (i.e num rows in embedding vector)\nmaxlen = 100 # max number of words in a question to use\n\nS_DROPOUT = 0.4\nDROPOUT = 0.1","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import os\nimport time\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nfrom tqdm import tqdm\nimport math\nfrom sklearn.model_selection import train_test_split\nfrom sklearn import metrics\nimport gc\n\nfrom keras.preprocessing.text import Tokenizer\nfrom keras.preprocessing.sequence import pad_sequences\nfrom keras.layers import Dense, Input, LSTM, Embedding, Dropout, Activation, CuDNNGRU, Conv1D, CuDNNLSTM, concatenate\nfrom keras.layers import Bidirectional, GlobalMaxPool1D, Dropout, SpatialDropout1D, GlobalAveragePooling1D, GlobalMaxPooling1D\nfrom keras.models import Model\nfrom keras import initializers, regularizers, constraints, optimizers, layers","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"train_df = pd.read_csv(\"../input/train.csv\")\ntest_df = pd.read_csv(\"../input/test.csv\")\nprint(\"Train shape : \",train_df.shape)\nprint(\"Test shape : \",test_df.shape)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ba5a1b8109dee2c9fbc628d5da4a7c3447d42fb8"},"cell_type":"code","source":"## split to train and val\ntrain_df, val_df = train_test_split(train_df, test_size=0.1, random_state=2018)\n\n## fill up the missing values\ntrain_X = train_df[\"question_text\"].fillna(\"_na_\").values\nval_X = val_df[\"question_text\"].fillna(\"_na_\").values\ntest_X = test_df[\"question_text\"].fillna(\"_na_\").values\n\n## Tokenize the sentences\ntokenizer = Tokenizer(num_words=max_features)\ntokenizer.fit_on_texts(list(train_X))\ntrain_X = tokenizer.texts_to_sequences(train_X)\nval_X = tokenizer.texts_to_sequences(val_X)\ntest_X = tokenizer.texts_to_sequences(test_X)\n\n## Pad the sentences \ntrain_X = pad_sequences(train_X, maxlen=maxlen)\nval_X = pad_sequences(val_X, maxlen=maxlen)\ntest_X = pad_sequences(test_X, maxlen=maxlen)\n\n## Get the target values\ntrain_y = train_df['target'].values\nval_y = val_df['target'].values","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"717f7dcd5ccf71e83d0f062e221c46db39845e4d"},"cell_type":"markdown","source":"**Glove Embeddings:**"},{"metadata":{"trusted":true,"_uuid":"23f130e80159bb1701e449e2e91199dbfff1f1d4"},"cell_type":"code","source":"EMBEDDING_FILE = '../input/embeddings/glove.840B.300d/glove.840B.300d.txt'\ndef get_coefs(word,*arr): return word, np.asarray(arr, dtype='float32')\nembeddings_index = dict(get_coefs(*o.split(\" \")) for o in open(EMBEDDING_FILE))\n\nall_embs = np.stack(embeddings_index.values())\nemb_mean,emb_std = all_embs.mean(), all_embs.std()\nembed_size = all_embs.shape[1]\n\nword_index = tokenizer.word_index\nnb_words = min(max_features, len(word_index))\nembedding_matrix_1 = np.random.normal(emb_mean, emb_std, (nb_words, embed_size))\nfor word, i in word_index.items():\n    if i >= max_features: continue\n    embedding_vector = embeddings_index.get(word)\n    if embedding_vector is not None: embedding_matrix_1[i] = embedding_vector\n\ndel embeddings_index; gc.collect() ","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"bc6bab22dd12a09378f4b8b159cb7a5d88a3e7c0"},"cell_type":"markdown","source":"**Wiki News FastText Embeddings:**"},{"metadata":{"trusted":true,"_uuid":"6f3d0fd28dd2b04eaccb732b96b872e5a223d962"},"cell_type":"code","source":"EMBEDDING_FILE = '../input/embeddings/wiki-news-300d-1M/wiki-news-300d-1M.vec'\ndef get_coefs(word,*arr): return word, np.asarray(arr, dtype='float32')\nembeddings_index = dict(get_coefs(*o.split(\" \")) for o in open(EMBEDDING_FILE) if len(o)>100)\n\nall_embs = np.stack(embeddings_index.values())\nemb_mean,emb_std = all_embs.mean(), all_embs.std()\nembed_size = all_embs.shape[1]\n\nword_index = tokenizer.word_index\nnb_words = min(max_features, len(word_index))\nembedding_matrix_2 = np.random.normal(emb_mean, emb_std, (nb_words, embed_size))\nfor word, i in word_index.items():\n    if i >= max_features: continue\n    embedding_vector = embeddings_index.get(word)\n    if embedding_vector is not None: embedding_matrix_2[i] = embedding_vector\n        \ndel embeddings_index; gc.collect()         ","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"4ca44ac68bf404b9c26e07fbcc9c8ac793e04510"},"cell_type":"markdown","source":"**Paragram Embeddings:**"},{"metadata":{"trusted":true,"_uuid":"25ec1aac4aedbf431a2d30de64030ce8e3203c18"},"cell_type":"code","source":"EMBEDDING_FILE = '../input/embeddings/paragram_300_sl999/paragram_300_sl999.txt'\ndef get_coefs(word,*arr): return word, np.asarray(arr, dtype='float32')\nembeddings_index = dict(get_coefs(*o.split(\" \")) for o in open(EMBEDDING_FILE, encoding=\"utf8\", errors='ignore') if len(o)>100)\n\nall_embs = np.stack(embeddings_index.values())\nemb_mean,emb_std = all_embs.mean(), all_embs.std()\nembed_size = all_embs.shape[1]\n\nword_index = tokenizer.word_index\nnb_words = min(max_features, len(word_index))\nembedding_matrix_3 = np.random.normal(emb_mean, emb_std, (nb_words, embed_size))\nfor word, i in word_index.items():\n    if i >= max_features: continue\n    embedding_vector = embeddings_index.get(word)\n    if embedding_vector is not None: embedding_matrix_3[i] = embedding_vector\n\ndel embeddings_index; gc.collect()         ","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"4474db5a07969ca0703c65c3f95c2b0f61c1468b"},"cell_type":"markdown","source":"**Word2vec Embeddings:**"},{"metadata":{"trusted":true,"_uuid":"9b83944a5ad66f4b5d60d57bed12b9c19035a19a"},"cell_type":"code","source":"# https://www.kaggle.com/strideradu/word2vec-and-gensim-go-go-go\nfrom gensim.models import KeyedVectors\n\nEMBEDDING_FILE = '../input/embeddings/GoogleNews-vectors-negative300/GoogleNews-vectors-negative300.bin'\nembeddings_index = KeyedVectors.load_word2vec_format(EMBEDDING_FILE, binary=True)\n\nword_index = tokenizer.word_index\nnb_words = min(max_features, len(word_index))\nembedding_matrix_4 = (np.random.rand(nb_words, embed_size) - 0.5) / 5.0\nfor word, i in word_index.items():\n    if i >= max_features: continue\n    if word in embeddings_index:\n        embedding_vector = embeddings_index.get_vector(word)\n        embedding_matrix_4[i] = embedding_vector\n        \ndel embeddings_index; gc.collect()         ","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"94e1be128208411c4d95a26610a4d58e66be013c"},"cell_type":"markdown","source":"** Combine :**"},{"metadata":{"trusted":true,"_uuid":"957c8187e76769e3165e900a9784fc4bb30ffca4"},"cell_type":"code","source":"embedding_matrix = np.concatenate((embedding_matrix_1, embedding_matrix_2, embedding_matrix_3, embedding_matrix_4), axis=1)  \ndel embedding_matrix_1, embedding_matrix_2, embedding_matrix_3, embedding_matrix_4\ngc.collect()\nnp.shape(embedding_matrix)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"77c68caa1c640b9e463a196cbca50304ae134d75"},"cell_type":"markdown","source":"**LSTM:**"},{"metadata":{"trusted":true,"_uuid":"ef7e53b15cf73d04747f1d225f25a3c37c1eef5f"},"cell_type":"code","source":"# https://www.kaggle.com/sudalairajkumar/a-look-at-different-embeddings\n# https://www.kaggle.com/strideradu/word2vec-and-gensim-go-go-go\n\ninp = Input(shape=(maxlen,))\nx = Embedding(max_features, embed_size * 4, weights=[embedding_matrix])(inp)\nx = SpatialDropout1D(S_DROPOUT)(x)\nx = Bidirectional(CuDNNGRU(128, return_sequences=True))(x)\navg_pool = GlobalAveragePooling1D()(x)\nmax_pool = GlobalMaxPooling1D()(x)\nconc = concatenate([avg_pool, max_pool])\nx = Dense(16, activation=\"relu\")(conc)\nx = Dropout(DROPOUT)(x)\nx = Dense(1, activation=\"sigmoid\")(x)\nmodel = Model(inputs=inp, outputs=x)\nmodel.compile(loss='binary_crossentropy', optimizer='adam', metrics=['accuracy'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"cc188f2787ea7b98d3a40953a95a5fc09ff2764d"},"cell_type":"code","source":"model.fit(train_X, train_y, batch_size=512, epochs=2, validation_data=(val_X, val_y))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"3432b48b6995e297fe33748c76174bd0e56c74a2"},"cell_type":"code","source":"pred_val_lstm_y = model.predict([val_X], batch_size=1024, verbose=1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"99cb9f6145da909bd7436e46d47547efc097499d"},"cell_type":"code","source":"pred_test_lstm_y = model.predict([test_X], batch_size=1024, verbose=1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"af087d21bdb4358701e31aded6b522accd5a8a64"},"cell_type":"code","source":"del all_embs, model, inp, x\nimport gc; gc.collect()\ntime.sleep(10)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"230dfd912022ba8a270766d3d5805d170afcfca6"},"cell_type":"markdown","source":"**CNN**"},{"metadata":{"trusted":true,"_uuid":"ddbc4f12d991399fa735066e2f0656b94b551602"},"cell_type":"code","source":"# https://www.kaggle.com/yekenot/2dcnn-textclassifier\nfrom keras.layers import Input, Embedding, Dense, Conv2D, MaxPool2D\nfrom keras.layers import Reshape, Flatten, Concatenate, Dropout, SpatialDropout1D\n\nfilter_sizes = [1,2,3,5]\nnum_filters = 36\n\ninp = Input(shape=(maxlen,))\nx = Embedding(max_features, embed_size * 4, weights=[embedding_matrix])(inp)\nx = SpatialDropout1D(S_DROPOUT)(x)\nx = Reshape((maxlen, embed_size * 4, 1))(x)\n\nmaxpool_pool = []\nfor i in range(len(filter_sizes)):\n    conv = Conv2D(num_filters, kernel_size=(filter_sizes[i], embed_size * 4),\n                                 kernel_initializer='he_normal', activation='elu')(x)\n    maxpool_pool.append(MaxPool2D(pool_size=(maxlen - filter_sizes[i] + 1, 1))(conv))\n\nz = Concatenate(axis=1)(maxpool_pool)   \nz = Flatten()(z)\nz = Dropout(DROPOUT)(z)\n\noutp = Dense(1, activation=\"sigmoid\")(z)\n\nmodel = Model(inputs=inp, outputs=outp)\nmodel.compile(loss='binary_crossentropy', optimizer='adam', metrics=['accuracy'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e5b2b647d5f65123db9889afe53fd0e5a86b2a9e"},"cell_type":"code","source":"model.fit(train_X, train_y, batch_size=512, epochs=2, validation_data=(val_X, val_y))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a5f010df43d7b6aeb7c3d040d1dd4ce839b6b4ca"},"cell_type":"code","source":"pred_val_cnn_y = model.predict([val_X], batch_size=1024, verbose=1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"688b1400e59b9ae1a321efd6d16e9065b353f49f"},"cell_type":"code","source":"pred_test_cnn_y = model.predict([test_X], batch_size=1024, verbose=1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"21260342b5b41fa161b664bcfed0da749802446e"},"cell_type":"code","source":"# del word_index, embedding_matrix, model, inp, x\nimport gc; gc.collect()\ntime.sleep(10)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"17e67a2a48ca69332c6e4edf2493f4c066c625ce"},"cell_type":"code","source":"pred_val_y = 0.6 * pred_val_lstm_y + 0.4 * pred_val_cnn_y  # two random numbers :)\npred_test_y = 0.6 * pred_test_lstm_y + 0.4 * pred_test_cnn_y \n\nthresholds = []\nfor thresh in np.arange(0.1, 0.501, 0.01):\n    thresh = np.round(thresh, 2)\n    res = metrics.f1_score(val_y, (pred_val_y > thresh).astype(int))\n    thresholds.append([thresh, res])\n    print(\"F1 score at threshold {0} is {1}\".format(thresh, res))\n    \nthresholds.sort(key=lambda x: x[1], reverse=True)\nbest_thresh = thresholds[0][0]\nprint(\"Best threshold: \", best_thresh)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"26e07c47be2330edb90ae6dd6b87711a71ac42e5"},"cell_type":"code","source":"pred_test_y = (pred_test_y > best_thresh).astype(int)\nout_df = pd.DataFrame({\"qid\":test_df[\"qid\"].values})\nout_df['prediction'] = pred_test_y\nout_df.to_csv(\"submission.csv\", index=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f6797ab73bdd5bdb8c8f6d80ec361c50a2b0f56f"},"cell_type":"markdown","source":"**References:**\n\n1. https://www.kaggle.com/jhoward/improved-lstm-baseline-glove-dropout"}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}