{"cells":[{"metadata":{"_uuid":"f1db88c4a899d3012619bf69361d7bdaf22f5f7e"},"cell_type":"markdown","source":"## Quroa Kaggle Challenge\n\nLatest Attempt at the Quora kaggle challenge.  See if older CNN n-gram can beat LSTM based solution\n\nFor each qid in the test set, you must predict whether the corresponding question_text is insincere (1) or not (0). Predictions should only be the integers 0 or 1.\n\n- https://www.kaggle.com/c/quora-insincere-questions-classification"},{"metadata":{"trusted":true,"_uuid":"002c934754d1e4f7f64ca18535a482c89dc11a48"},"cell_type":"code","source":"import os\nimport time\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nfrom tqdm import tqdm\nimport math\nfrom sklearn.model_selection import train_test_split\nfrom sklearn import metrics\n\nfrom keras.preprocessing.text import Tokenizer\nfrom keras.preprocessing.sequence import pad_sequences\nfrom keras.preprocessing.text import Tokenizer\nfrom keras.preprocessing.sequence import pad_sequences\nfrom keras.layers import Dense, Input, LSTM, Embedding, Dropout, Activation, CuDNNGRU, Conv1D\nfrom keras.layers import Bidirectional, GlobalMaxPool1D\nfrom keras.models import Model\nfrom keras import initializers, regularizers, constraints, optimizers, layers\nfrom keras.models import Sequential\nfrom keras.layers import Dense\nfrom keras.layers import Flatten\nfrom keras.layers import Embedding\nfrom keras.layers.convolutional import Conv1D\nfrom keras.layers.convolutional import MaxPooling1D\nfrom keras.layers.merge import concatenate\n\nSEED = 2018\nimport tensorflow as tf\nnp.random.seed(SEED)\ntf.set_random_seed(SEED)\n\nfrom tqdm import tqdm\ntqdm.pandas()\n\nimport gc,re\n\nimport warnings\nwarnings.filterwarnings('ignore')\nimport numpy as np","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"7e60ba7031b849987ee9c7ac9424b2086927cef8"},"cell_type":"code","source":"#do a clean up\n#del word_index, embeddings_index, all_embs, embedding_matrix, model\n#import gc; gc.collect()\n#time.sleep(10)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b01e4c2d0ddee48070bd201e99682ee36492c5f7"},"cell_type":"code","source":"#read in the data\ntrain_df = pd.read_csv(\"../input/train.csv\")\ntest_df = pd.read_csv(\"../input/test.csv\")\nprint(\"Train shape : \",train_df.shape)\nprint(\"Test shape : \",test_df.shape)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"941bb3bb9c046795ac4ced291263f0d17379c893"},"cell_type":"code","source":"#lets look at the data\ntrain_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"5cfcbc3afbf24c839840f4f270e6aeca13d62e17"},"cell_type":"code","source":"def clean_text(x):\n    x = str(x)\n    for punct in \"/-'\":\n        x = x.replace(punct, ' ')\n    for punct in '&':\n        x = x.replace(punct, f' {punct} ')\n    for punct in '?!.,\"#$%\\'()*+-/:;<=>@[\\\\]^_`{|}~' + '“”’':\n        x = x.replace(punct, '')\n    return x\n\ndef clean_numbers(x):\n    x = re.sub('[0-9]{5,}', '#####', x)\n    x = re.sub('[0-9]{4}', '####', x)\n    x = re.sub('[0-9]{3}', '###', x)\n    x = re.sub('[0-9]{2}', '##', x)\n    return x\n\ndef _get_mispell(mispell_dict):\n    mispell_re = re.compile('(%s)' % '|'.join(mispell_dict.keys()))\n    return mispell_dict, mispell_re\n\nmispell_dict = {'colour':'color',\n                'centre':'center',\n                'didnt':'did not',\n                'doesnt':'does not',\n                'isnt':'is not',\n                'shouldnt':'should not',\n                'favourite':'favorite',\n                'travelling':'traveling',\n                'counselling':'counseling',\n                'theatre':'theater',\n                'cancelled':'canceled',\n                'labour':'labor',\n                'organisation':'organization',\n                'wwii':'world war 2',\n                'citicise':'criticize',\n                'instagram': 'social medium',\n                'whatsapp': 'social medium',\n                'snapchat': 'social medium'\n                }\nmispellings, mispellings_re = _get_mispell(mispell_dict)\n\ndef replace_typical_misspell(text):\n    def replace(match):\n        return mispellings[match.group(0)]\n    return mispellings_re.sub(replace, text)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"5fb37862768a961f337846c52f579aee97149004"},"cell_type":"code","source":"# Clean the text\ntrain_df[\"question_text\"] = train_df[\"question_text\"].progress_apply(lambda x: clean_text(x))\ntest_df[\"question_text\"] = test_df[\"question_text\"].progress_apply(lambda x: clean_text(x))\n\n# Clean numbers\ntrain_df[\"question_text\"] = train_df[\"question_text\"].progress_apply(lambda x: clean_numbers(x))\ntest_df[\"question_text\"] = test_df[\"question_text\"].progress_apply(lambda x: clean_numbers(x))\n\n# Clean speelings\ntrain_df[\"question_text\"] = train_df[\"question_text\"].progress_apply(lambda x: replace_typical_misspell(x))\ntest_df[\"question_text\"] = test_df[\"question_text\"].progress_apply(lambda x: replace_typical_misspell(x))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f956fb69167a41cbe288f5eab9ccf49392fa117a"},"cell_type":"code","source":"train_df.head()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"e567fbf8fdb2d2f703a1d694a93ee49829c866c0"},"cell_type":"markdown","source":"Next steps are as follows:\n * Split the training dataset into train and val sample - cross val too expensive\n * Fill up the missing values in the text column with '_na_'\n * Tokenize the text column and convert them to vector sequences\n * Pad the sequence as needed - if the number of words in the text is greater than 'max_len' trunacate them to 'max_len' or if the number of words in the text is lesser than 'max_len' add zeros for remaining values."},{"metadata":{"trusted":true,"_uuid":"96b630906cd59024ca3d31ad1bd9048bbfe541c1"},"cell_type":"code","source":"## split to train and val\ntrain_df, val_df = train_test_split(train_df, test_size=0.1, random_state=2018)\n\n## some config values \nembed_size = 300 # how big is each word vector\nmax_features = 50000 # how many unique words to use (i.e num rows in embedding vector)\nmaxlen = 100 # max number of words in a question to use\n\n## fill up the missing values\ntrain_X = train_df[\"question_text\"].fillna(\"_na_\").values\nval_X = val_df[\"question_text\"].fillna(\"_na_\").values\ntest_X = test_df[\"question_text\"].fillna(\"_na_\").values","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"9a43a9f2d39ab69632b6485006feb47a82b3f138"},"cell_type":"code","source":"## Tokenize the sentences\ntokenizer = Tokenizer(num_words=max_features)\ntokenizer.fit_on_texts(list(train_X))\n#produce a list of lists- each list is a integer representation of each word in the sentence\ntrain_X = tokenizer.texts_to_sequences(train_X)\n#print((train_X[:10]))\n\n\nval_X = tokenizer.texts_to_sequences(val_X)\ntest_X = tokenizer.texts_to_sequences(test_X)\n\n## Pad the sentences \ntrain_X = pad_sequences(train_X, maxlen=maxlen)\nval_X = pad_sequences(val_X, maxlen=maxlen)\ntest_X = pad_sequences(test_X, maxlen=maxlen)\n\n## Get the target values\ntrain_y = train_df['target'].values\nval_y = val_df['target'].values","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"88cd2a20d46972b3c12d4da114162bb79b77cfb6"},"cell_type":"code","source":"EMBEDDING_FILE = '../input/embeddings/glove.840B.300d/glove.840B.300d.txt'\ndef get_coefs(word,*arr): return word, np.asarray(arr, dtype='float32')\nembeddings_index = dict(get_coefs(*o.split(\" \")) for o in open(EMBEDDING_FILE,errors='ignore'))\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"92ee5a56f8eccd6ac7cb45fd66c653da84da27b0"},"cell_type":"code","source":"all_embs = np.stack(embeddings_index.values())\nemb_mean,emb_std = all_embs.mean(), all_embs.std()\nembed_size = all_embs.shape[1]\n\nword_index = tokenizer.word_index\nnb_words = min(max_features, len(word_index))\nprint(nb_words)\n\nword_index = tokenizer.word_index\nnb_words = min(max_features, len(word_index))\nembedding_matrix = np.random.normal(emb_mean, emb_std, (nb_words, embed_size))\nfor word, i in word_index.items():\n    if i >= max_features: continue\n    embedding_vector = embeddings_index.get(word)\n    if embedding_vector is not None: embedding_matrix[i] = embedding_vector","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"de9bcd06cb7e62882d5e7902b6a0612a570efcdb"},"cell_type":"code","source":"train_X.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"452b2708bafc0cc3b9052744e117788a15eacf58"},"cell_type":"code","source":"#inp = Input(shape=(maxlen,))\n#x = Embedding(max_features, embed_size, weights=[embedding_matrix])(inp)\n#x = Bidirectional(CuDNNGRU(64, return_sequences=True))(x)\n#x = GlobalMaxPool1D()(x)\n#x = Dense(16, activation=\"relu\")(x)\n#x = Dropout(0.1)(x)\n#x = Dense(1, activation=\"sigmoid\")(x)\n#model = Model(inputs=inp, outputs=x)\n#model.compile(loss='binary_crossentropy', optimizer='adam', metrics=['accuracy'])\n#print(model.summary())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"643d304c6a87163acfc4da5932e1068b2bddc510"},"cell_type":"code","source":"# define the model\ndef define_model(length=maxlen, vocab_size=max_features):\n    # channel 1\n    inputs1 = Input(shape=(length,))\n    embedding1 = Embedding(vocab_size, embed_size,weights=[embedding_matrix])(inputs1)\n    conv1 = Conv1D(filters=32, kernel_size=4, activation='relu')(embedding1)\n    drop1 = Dropout(0.5)(conv1)\n    pool1 = MaxPooling1D(pool_size=2)(drop1)\n    flat1 = Flatten()(pool1)\n    # channel 2\n    inputs2 = Input(shape=(length,))\n    embedding2 = Embedding(vocab_size, embed_size,weights=[embedding_matrix])(inputs2)\n    conv2 = Conv1D(filters=32, kernel_size=6, activation='relu')(embedding2)\n    drop2 = Dropout(0.5)(conv2)\n    pool2 = MaxPooling1D(pool_size=2)(drop2)\n    flat2 = Flatten()(pool2)\n    # channel 3\n    inputs3 = Input(shape=(length,))\n    embedding3 = Embedding(vocab_size, embed_size,weights=[embedding_matrix])(inputs3)\n    conv3 = Conv1D(filters=32, kernel_size=8, activation='relu')(embedding3)\n    drop3 = Dropout(0.5)(conv3)\n    pool3 = MaxPooling1D(pool_size=2)(drop3)\n    flat3 = Flatten()(pool3)\n    # merge\n    merged = concatenate([flat1, flat2, flat3])\n    # interpretation\n    dense1 = Dense(10, activation='relu')(merged)\n    outputs = Dense(1, activation='sigmoid')(dense1)\n    model = Model(inputs=[inputs1, inputs2, inputs3], outputs=outputs)\n    # compile\n    model.compile(loss='binary_crossentropy', optimizer='adam', metrics=['accuracy'])\n    # summarize\n    model.summary()\n    #plot_model(model, show_shapes=True, to_file='model.png')\n    return model","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"8b8ebe22269db36652b71ed45a0f47f8f362c438"},"cell_type":"code","source":"model = define_model(maxlen, max_features)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"473b300af87aba740e01323f350097f0265d0ebb"},"cell_type":"code","source":"model.fit([train_X,train_X,train_X], train_y, batch_size=512, epochs=2,validation_data=([val_X,val_X,val_X], val_y),verbose=2)\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d2f114e2332786d5ff58fa51788938042d90f8c1"},"cell_type":"code","source":"#apply to validation set\npred_glove_val_y = model.predict([val_X,val_X,val_X], batch_size=512, verbose=1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"10238f62f782697bc9870d5122bf97c25a156fcb"},"cell_type":"code","source":"#look for a better threshold\nfor thresh in np.arange(0.1, 0.501, 0.01):\n    thresh = np.round(thresh, 2)\n    print(\"F1 score at threshold {0} is {1}\".format(thresh, metrics.f1_score(val_y, (pred_glove_val_y>thresh).astype(int))))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"fe55ff3bd7c303479feeda6aa100eb558b1a3263"},"cell_type":"code","source":"pred_glove_test_y = model.predict([test_X,test_X,test_X], batch_size=512, verbose=1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"171f323089ea48c67df89fff3a1223d206fd2f81"},"cell_type":"code","source":"pred_glove_test_y","execution_count":null,"outputs":[]},{"metadata":{"scrolled":true,"trusted":true,"_uuid":"20dab6b8de766c50fefbbfa25d9419e9d9a8fee8"},"cell_type":"code","source":"#do a clean up\ndel word_index, embeddings_index, all_embs, embedding_matrix, model\nimport gc; gc.collect()\ntime.sleep(10)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"18672e23e4d9f64da3ab2b96f2ce5f8ca2ee57b7"},"cell_type":"markdown","source":"**Wiki News FastText Embeddings:**\n\nNow let us use the FastText embeddings trained on Wiki News corpus in place of Glove embeddings and rebuild the model."},{"metadata":{"trusted":true,"_uuid":"4b782512670c6f302f6de30983f5032eeda3b1ec"},"cell_type":"code","source":"EMBEDDING_FILE = '../input/embeddings/wiki-news-300d-1M/wiki-news-300d-1M.vec'\ndef get_coefs(word,*arr): return word, np.asarray(arr, dtype='float32')\nembeddings_index = dict(get_coefs(*o.split(\" \")) for o in open(EMBEDDING_FILE,errors='ignore') if len(o)>100)\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f8a2b2a19b0b3d852d54965328ea137f6f0d7fdd"},"cell_type":"code","source":"all_embs = np.stack(embeddings_index.values())\nemb_mean,emb_std = all_embs.mean(), all_embs.std()\nembed_size = all_embs.shape[1]\nprint(all_embs.shape)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"60265868bc333f984783ef16905c0c0c4c442564"},"cell_type":"code","source":"word_index = tokenizer.word_index\nprint(len(word_index))\nnb_words = min(max_features, len(word_index))\nembedding_matrix = np.random.normal(emb_mean, emb_std, (nb_words, embed_size))\nprint(embedding_matrix.shape)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f6843ecbaf097bee5b338de6404fdbcc45e1237d"},"cell_type":"code","source":"for word, i in word_index.items():\n    if i >= max_features: continue\n    embedding_vector = embeddings_index.get(word)\n    if embedding_vector is not None: embedding_matrix[i] = embedding_vector\n\n#len(word_index.items())\n#print(len(embedding_vector))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"4920dc3e820407617df0a79e191d95cb579cbc3b"},"cell_type":"code","source":"# define the model\ndef define_model(length=maxlen, vocab_size=max_features):\n    # channel 1\n    inputs1 = Input(shape=(length,))\n    embedding1 = Embedding(vocab_size, embed_size,weights=[embedding_matrix])(inputs1)\n    conv1 = Conv1D(filters=32, kernel_size=4, activation='relu')(embedding1)\n    drop1 = Dropout(0.5)(conv1)\n    pool1 = MaxPooling1D(pool_size=2)(drop1)\n    flat1 = Flatten()(pool1)\n    # channel 2\n    inputs2 = Input(shape=(length,))\n    embedding2 = Embedding(vocab_size, embed_size,weights=[embedding_matrix])(inputs2)\n    conv2 = Conv1D(filters=32, kernel_size=6, activation='relu')(embedding2)\n    drop2 = Dropout(0.5)(conv2)\n    pool2 = MaxPooling1D(pool_size=2)(drop2)\n    flat2 = Flatten()(pool2)\n    # channel 3\n    inputs3 = Input(shape=(length,))\n    embedding3 = Embedding(vocab_size, embed_size,weights=[embedding_matrix])(inputs3)\n    conv3 = Conv1D(filters=32, kernel_size=8, activation='relu')(embedding3)\n    drop3 = Dropout(0.5)(conv3)\n    pool3 = MaxPooling1D(pool_size=2)(drop3)\n    flat3 = Flatten()(pool3)\n    # merge\n    merged = concatenate([flat1, flat2, flat3])\n    # interpretation\n    dense1 = Dense(10, activation='relu')(merged)\n    outputs = Dense(1, activation='sigmoid')(dense1)\n    model = Model(inputs=[inputs1, inputs2, inputs3], outputs=outputs)\n    # compile\n    model.compile(loss='binary_crossentropy', optimizer='adam', metrics=['accuracy'])\n    # summarize\n    model.summary()\n    #plot_model(model, show_shapes=True, to_file='model.png')\n    return model\nmodel = define_model(maxlen, max_features)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ba0c597513ec24847f2130ada2b77573220d4759"},"cell_type":"code","source":"model.fit([train_X,train_X,train_X], train_y, batch_size=512, epochs=2,validation_data=([val_X,val_X,val_X], val_y),verbose=2)\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"0ba0c3c264f07ccb986a109b7b98c95f3deb5f68"},"cell_type":"code","source":"pred_fasttext_val_y = model.predict([val_X,val_X,val_X], batch_size=128, verbose=1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"cc7a751e39bd747d5a5b340d04e1594099de2700"},"cell_type":"code","source":"for thresh in np.arange(0.1, 0.501, 0.01):\n    thresh = np.round(thresh, 2)\n    print(\"F1 score at threshold {0} is {1}\".format(thresh, metrics.f1_score(val_y, (pred_fasttext_val_y>thresh).astype(int))))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a1688bd4f3b012cbc516537ea33c0ce9e26c3abc"},"cell_type":"code","source":"pred_fasttext_test_y = model.predict([test_X,test_X,test_X], batch_size=512, verbose=1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"51774ea5d0d8776fa9af482e80d3f1c96cad8e39"},"cell_type":"code","source":"#del  embeddings_index, all_embs, embedding_matrix, model, x\n#import gc; gc.collect()\n#time.sleep(10)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"e5d87f0e548b0333e51c8020461c72e4cfb49c73"},"cell_type":"markdown","source":"**Paragram Embeddings:**\n\nIn this section, we can use the paragram embeddings and build the model and make predictions."},{"metadata":{"trusted":true,"_uuid":"b1fa6961917806a7614f5cab0c0b37e27552c686"},"cell_type":"code","source":"EMBEDDING_FILE = '../input/embeddings/paragram_300_sl999/paragram_300_sl999.txt'\ndef get_coefs(word,*arr): return word, np.asarray(arr, dtype='float32')\nembeddings_index = dict(get_coefs(*o.split(\" \")) for o in open(EMBEDDING_FILE, encoding=\"utf8\", errors='ignore') if len(o)>100)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"51d3b50889d1b8be9488db2fe76faabfd280beba"},"cell_type":"code","source":"\nall_embs = np.stack(embeddings_index.values())\nemb_mean,emb_std = all_embs.mean(), all_embs.std()\nembed_size = all_embs.shape[1]\n\nword_index = tokenizer.word_index\nnb_words = min(max_features, len(word_index))\nembedding_matrix = np.random.normal(emb_mean, emb_std, (nb_words, embed_size))\nfor word, i in word_index.items():\n    if i >= max_features: continue\n    embedding_vector = embeddings_index.get(word)\n    if embedding_vector is not None: embedding_matrix[i] = embedding_vector\n        ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e69f1e094b121bcf45583ecea296a2c46053aa51"},"cell_type":"code","source":"# define the model\ndef define_model(length=maxlen, vocab_size=max_features):\n    # channel 1\n    inputs1 = Input(shape=(length,))\n    embedding1 = Embedding(vocab_size, embed_size,weights=[embedding_matrix])(inputs1)\n    conv1 = Conv1D(filters=32, kernel_size=4, activation='relu')(embedding1)\n    drop1 = Dropout(0.5)(conv1)\n    pool1 = MaxPooling1D(pool_size=2)(drop1)\n    flat1 = Flatten()(pool1)\n    # channel 2\n    inputs2 = Input(shape=(length,))\n    embedding2 = Embedding(vocab_size, embed_size,weights=[embedding_matrix])(inputs2)\n    conv2 = Conv1D(filters=32, kernel_size=6, activation='relu')(embedding2)\n    drop2 = Dropout(0.5)(conv2)\n    pool2 = MaxPooling1D(pool_size=2)(drop2)\n    flat2 = Flatten()(pool2)\n    # channel 3\n    inputs3 = Input(shape=(length,))\n    embedding3 = Embedding(vocab_size, embed_size,weights=[embedding_matrix])(inputs3)\n    conv3 = Conv1D(filters=32, kernel_size=8, activation='relu')(embedding3)\n    drop3 = Dropout(0.5)(conv3)\n    pool3 = MaxPooling1D(pool_size=2)(drop3)\n    flat3 = Flatten()(pool3)\n    # merge\n    merged = concatenate([flat1, flat2, flat3])\n    # interpretation\n    dense1 = Dense(10, activation='relu')(merged)\n    outputs = Dense(1, activation='sigmoid')(dense1)\n    model = Model(inputs=[inputs1, inputs2, inputs3], outputs=outputs)\n    # compile\n    model.compile(loss='binary_crossentropy', optimizer='adam', metrics=['accuracy'])\n    # summarize\n    model.summary()\n    #plot_model(model, show_shapes=True, to_file='model.png')\n    return model\nmodel = define_model(maxlen, max_features)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"4b78fa1e6b2c2a75e87ecac8230f66f8915b9c75"},"cell_type":"code","source":"model.fit([train_X,train_X,train_X], train_y, batch_size=512, epochs=2,validation_data=([val_X,val_X,val_X], val_y),verbose=2)\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"877851735e0436bbd62ac392227483f71a924ab3"},"cell_type":"code","source":"pred_paragram_val_y = model.predict([val_X,val_X,val_X], batch_size=1024, verbose=1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"606e8c8282a319986491d08a4e915576a6d62032"},"cell_type":"code","source":"for thresh in np.arange(0.1, 0.501, 0.01):\n    thresh = np.round(thresh, 2)\n    print(\"F1 score at threshold {0} is {1}\".format(thresh, metrics.f1_score(val_y, (pred_paragram_val_y>thresh).astype(int))))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"1e4b1e9a7b9989c5b0d3a093af5db7047be66a5b"},"cell_type":"code","source":"pred_paragram_test_y = model.predict([test_X,test_X,test_X], batch_size=512, verbose=1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"86f81f703dc25b580a5c0b885a67d632caf5198f"},"cell_type":"code","source":"#del word_index, embeddings_index, all_embs, embedding_matrix, model, inp, x\n#import gc; gc.collect()\n#time.sleep(10)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"4bedf1d5a3baa182970f28ad4f73696a5c2619e1"},"cell_type":"code","source":"pred_val_y = 0.33*pred_glove_val_y + 0.33*pred_fasttext_val_y + 0.33*pred_paragram_val_y ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"2f70cb3eb7c9d4c1d876ac310b33c986bef6a50d"},"cell_type":"code","source":"for thresh in np.arange(0.1, 0.501, 0.01):\n    thresh = np.round(thresh, 2)\n    print(\"F1 score at threshold {0} is {1}\".format(thresh, metrics.f1_score(val_y, (pred_val_y>thresh).astype(int))))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c9083981e962cd48b8634ea279ea8f2f4da1601d"},"cell_type":"code","source":"pred_test_y = 0.33*pred_glove_test_y + 0.33*pred_fasttext_test_y + 0.33*pred_paragram_test_y\npred_test_y = (pred_test_y>0.36).astype(int)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"78499e54773eca9dc69d45152284cd7f8f9bc97c"},"cell_type":"code","source":"out_df = pd.DataFrame({\"qid\":test_df[\"qid\"].values})\nout_df['prediction'] = pred_test_y\nout_df.to_csv(\"submission.csv\", index=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c4e91f0d2335aadfaefb94d93ebdd56127983a1a"},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}