{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import os\nimport time\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nfrom tqdm import tqdm\nimport math\nfrom sklearn.model_selection import train_test_split\nfrom sklearn import metrics\n\nfrom keras.preprocessing.text import Tokenizer\nfrom keras.preprocessing.sequence import pad_sequences\nfrom keras.layers import Dense, Input, LSTM, Embedding, Dropout, Activation, CuDNNGRU, Conv1D\nfrom keras.layers import Bidirectional, GlobalMaxPool1D\nfrom keras.models import Model\nfrom keras import initializers, regularizers, constraints, optimizers, layers","execution_count":2,"outputs":[{"output_type":"stream","text":"Using TensorFlow backend.\n","name":"stderr"}]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"train = pd.read_csv(\"../input/quora-insincere-questions-classification/train.csv\")\ntest_df = pd.read_csv(\"../input/quora-insincere-questions-classification/test.csv\")\n# print(\"Train shape : \",train_df.shape)\nprint(\"Test shape : \",test_df.shape)","execution_count":10,"outputs":[{"output_type":"stream","text":"Test shape :  (375806, 2)\n","name":"stdout"}]},{"metadata":{"trusted":true},"cell_type":"code","source":"train.head()","execution_count":1,"outputs":[{"output_type":"error","ename":"NameError","evalue":"name 'train' is not defined","traceback":["\u001b[0;31m---------------------------------------------------------------------------\u001b[0m","\u001b[0;31mNameError\u001b[0m                                 Traceback (most recent call last)","\u001b[0;32m<ipython-input-1-3b77fa18a747>\u001b[0m in \u001b[0;36m<module>\u001b[0;34m()\u001b[0m\n\u001b[0;32m----> 1\u001b[0;31m \u001b[0mtrain\u001b[0m\u001b[0;34m.\u001b[0m\u001b[0mhead\u001b[0m\u001b[0;34m(\u001b[0m\u001b[0;34m)\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n\u001b[0m","\u001b[0;31mNameError\u001b[0m: name 'train' is not defined"]}]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df = pd.read_csv(\"../input/quora1/X_train (1).csv\")\nval_df = pd.read_csv(\"../input/quora1/X_val .csv\")\nprint(\"Train shape : \",train_df.shape)\nprint(\"Test shape : \",val_df.shape)","execution_count":5,"outputs":[{"output_type":"stream","text":"Train shape :  (162532, 9)\nTest shape :  (261225, 9)\n","name":"stdout"}]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df.head()\n# train_df['target'].value_counts()\n","execution_count":9,"outputs":[{"output_type":"execute_result","execution_count":9,"data":{"text/plain":"   Unnamed: 0  ...   target\n0      330096  ...        1\n1      148377  ...        1\n2      605482  ...        1\n3      278650  ...        1\n4      669320  ...        1\n\n[5 rows x 9 columns]","text/html":"<div>\n<style scoped>\n    .dataframe tbody tr th:only-of-type {\n        vertical-align: middle;\n    }\n\n    .dataframe tbody tr th {\n        vertical-align: top;\n    }\n\n    .dataframe thead th {\n        text-align: right;\n    }\n</style>\n<table border=\"1\" class=\"dataframe\">\n  <thead>\n    <tr style=\"text-align: right;\">\n      <th></th>\n      <th>Unnamed: 0</th>\n      <th>question_text</th>\n      <th>num_words</th>\n      <th>num_unique_words</th>\n      <th>num_chars</th>\n      <th>num_words_upper</th>\n      <th>num_words_title</th>\n      <th>mean_word_len</th>\n      <th>target</th>\n    </tr>\n  </thead>\n  <tbody>\n    <tr>\n      <th>0</th>\n      <td>330096</td>\n      <td>Was it really an \"accident\" that Buddy, the Cl...</td>\n      <td>41</td>\n      <td>37</td>\n      <td>236</td>\n      <td>0</td>\n      <td>5</td>\n      <td>4.780488</td>\n      <td>1</td>\n    </tr>\n    <tr>\n      <th>1</th>\n      <td>148377</td>\n      <td>Why is Quora full of self-obsessed and narciss...</td>\n      <td>9</td>\n      <td>9</td>\n      <td>59</td>\n      <td>0</td>\n      <td>2</td>\n      <td>5.666667</td>\n      <td>1</td>\n    </tr>\n    <tr>\n      <th>2</th>\n      <td>605482</td>\n      <td>How does Jimmy Wales feel about Wikipedia stat...</td>\n      <td>16</td>\n      <td>16</td>\n      <td>105</td>\n      <td>0</td>\n      <td>4</td>\n      <td>5.625000</td>\n      <td>1</td>\n    </tr>\n    <tr>\n      <th>3</th>\n      <td>278650</td>\n      <td>Is it true that Quora represents the Jews’ com...</td>\n      <td>9</td>\n      <td>9</td>\n      <td>53</td>\n      <td>0</td>\n      <td>3</td>\n      <td>5.000000</td>\n      <td>1</td>\n    </tr>\n    <tr>\n      <th>4</th>\n      <td>669320</td>\n      <td>How common is it when a man publicly disagrees...</td>\n      <td>25</td>\n      <td>22</td>\n      <td>146</td>\n      <td>0</td>\n      <td>1</td>\n      <td>4.880000</td>\n      <td>1</td>\n    </tr>\n  </tbody>\n</table>\n</div>"},"metadata":{}}]},{"metadata":{"trusted":true},"cell_type":"code","source":"## split to train and val\n# train_df, val_df = train_test_split(train_df, test_size=0.1, random_state=2018)\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"## some config values \nembed_size = 300 # how big is each word vector\nmax_features = 50000 # how many unique words to use (i.e num rows in embedding vector)\nmaxlen = 100 # max number of words in a question to use\n","execution_count":8,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"## fill up the missing values\ntrain_X = train_df[\"question_text\"].fillna(\"_na_\").values\nval_X = val_df[\"question_text\"].fillna(\"_na_\").values\ntest_X = test_df[\"question_text\"].fillna(\"_na_\").values\n","execution_count":11,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"## Tokenize the sentences\ntokenizer = Tokenizer(num_words=max_features)\ntokenizer.fit_on_texts(list(train_X))\ntrain_X = tokenizer.texts_to_sequences(train_X)\nval_X = tokenizer.texts_to_sequences(val_X)\ntest_X = tokenizer.texts_to_sequences(test_X)\n","execution_count":12,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"## Pad the sentences \ntrain_X = pad_sequences(train_X, maxlen=maxlen)\nval_X = pad_sequences(val_X, maxlen=maxlen)\ntest_X = pad_sequences(test_X, maxlen=maxlen)","execution_count":13,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"## Get the target values\ntrain_y = train_df['target'].values\nval_y = val_df['target'].values","execution_count":14,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# del model, inp, x\n# import gc; gc.collect()\n# time.sleep(10)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"!ls ../input/embeddings/","execution_count":15,"outputs":[{"output_type":"stream","text":"ls: cannot access '../input/embeddings/': No such file or directory\r\n","name":"stdout"}]},{"metadata":{"trusted":true},"cell_type":"code","source":"EMBEDDING_FILE = '../input/quora-insincere-questions-classification/embeddings/glove.840B.300d/glove.840B.300d.txt'\ndef get_coefs(word,*arr): return word, np.asarray(arr, dtype='float32')\nembeddings_index = dict(get_coefs(*o.split(\" \")) for o in open(EMBEDDING_FILE))\n\nall_embs = np.stack(embeddings_index.values())\nemb_mean,emb_std = all_embs.mean(), all_embs.std()\nembed_size = all_embs.shape[1]\n\nword_index = tokenizer.word_index\nnb_words = min(max_features, len(word_index))\nembedding_matrix = np.random.normal(emb_mean, emb_std, (nb_words, embed_size))\nfor word, i in word_index.items():\n    if i >= max_features: continue\n    embedding_vector = embeddings_index.get(word)\n    if embedding_vector is not None: embedding_matrix[i] = embedding_vector\n        \ninp = Input(shape=(maxlen,))\nx = Embedding(max_features, embed_size, weights=[embedding_matrix])(inp)\nx = Bidirectional(CuDNNGRU(64, return_sequences=True))(x)\nx = GlobalMaxPool1D()(x)\nx = Dense(16, activation=\"relu\")(x)\nx = Dropout(0.1)(x)\nx = Dense(1, activation=\"sigmoid\")(x)\nmodel = Model(inputs=inp, outputs=x)\nmodel.compile(loss='binary_crossentropy', optimizer='adam', metrics=['accuracy'])\nprint(model.summary())","execution_count":18,"outputs":[{"output_type":"stream","text":"/opt/conda/lib/python3.6/site-packages/ipykernel_launcher.py:5: FutureWarning: arrays to stack must be passed as a \"sequence\" type such as list or tuple. Support for non-sequence iterables such as generators is deprecated as of NumPy 1.16 and will raise an error in the future.\n  \"\"\"\n","name":"stderr"},{"output_type":"stream","text":"WARNING:tensorflow:From /opt/conda/lib/python3.6/site-packages/tensorflow/python/framework/op_def_library.py:263: colocate_with (from tensorflow.python.framework.ops) is deprecated and will be removed in a future version.\nInstructions for updating:\nColocations handled automatically by placer.\nWARNING:tensorflow:From /opt/conda/lib/python3.6/site-packages/keras/backend/tensorflow_backend.py:3445: calling dropout (from tensorflow.python.ops.nn_ops) with keep_prob is deprecated and will be removed in a future version.\nInstructions for updating:\nPlease use `rate` instead of `keep_prob`. Rate should be set to `rate = 1 - keep_prob`.\n_________________________________________________________________\nLayer (type)                 Output Shape              Param #   \n=================================================================\ninput_1 (InputLayer)         (None, 100)               0         \n_________________________________________________________________\nembedding_1 (Embedding)      (None, 100, 300)          15000000  \n_________________________________________________________________\nbidirectional_1 (Bidirection (None, 100, 128)          140544    \n_________________________________________________________________\nglobal_max_pooling1d_1 (Glob (None, 128)               0         \n_________________________________________________________________\ndense_1 (Dense)              (None, 16)                2064      \n_________________________________________________________________\ndropout_1 (Dropout)          (None, 16)                0         \n_________________________________________________________________\ndense_2 (Dense)              (None, 1)                 17        \n=================================================================\nTotal params: 15,142,625\nTrainable params: 15,142,625\nNon-trainable params: 0\n_________________________________________________________________\nNone\n","name":"stdout"}]},{"metadata":{"trusted":true},"cell_type":"code","source":"model.fit(train_X, train_y, batch_size=512, epochs=2, validation_data=(val_X, val_y))\n","execution_count":19,"outputs":[{"output_type":"stream","text":"WARNING:tensorflow:From /opt/conda/lib/python3.6/site-packages/tensorflow/python/ops/math_ops.py:3066: to_int32 (from tensorflow.python.ops.math_ops) is deprecated and will be removed in a future version.\nInstructions for updating:\nUse tf.cast instead.\nWARNING:tensorflow:From /opt/conda/lib/python3.6/site-packages/tensorflow/python/ops/math_grad.py:102: div (from tensorflow.python.ops.math_ops) is deprecated and will be removed in a future version.\nInstructions for updating:\nDeprecated in favor of operator or tf.math.divide.\nTrain on 162532 samples, validate on 261225 samples\nEpoch 1/2\n162532/162532 [==============================] - 21s 127us/step - loss: 0.2984 - acc: 0.8787 - val_loss: 0.1850 - val_acc: 0.9233\nEpoch 2/2\n162532/162532 [==============================] - 18s 108us/step - loss: 0.2216 - acc: 0.9170 - val_loss: 0.1840 - val_acc: 0.9227\n","name":"stdout"},{"output_type":"execute_result","execution_count":19,"data":{"text/plain":"<keras.callbacks.History at 0x7fcbda98b1d0>"},"metadata":{}}]},{"metadata":{"trusted":true},"cell_type":"code","source":"pred_glove_val_y = model.predict([val_X], batch_size=1024, verbose=1)\nfor thresh in np.arange(0.1, 0.501, 0.01):\n    thresh = np.round(thresh, 2)\n    print(\"F1 score at threshold {0} is {1}\".format(thresh, metrics.f1_score(val_y, (pred_glove_val_y>thresh).astype(int))))","execution_count":20,"outputs":[{"output_type":"stream","text":"261225/261225 [==============================] - 4s 17us/step\nF1 score at threshold 0.1 is 0.404996627406216\nF1 score at threshold 0.11 is 0.4146860777686105\nF1 score at threshold 0.12 is 0.4231738035264484\nF1 score at threshold 0.13 is 0.43121829175064286\nF1 score at threshold 0.14 is 0.43911360639700564\nF1 score at threshold 0.15 is 0.4463179777389597\nF1 score at threshold 0.16 is 0.45283905648738354\nF1 score at threshold 0.17 is 0.45897822420560885\nF1 score at threshold 0.18 is 0.464845702858701\nF1 score at threshold 0.19 is 0.4708216081561102\nF1 score at threshold 0.2 is 0.4762929354350843\nF1 score at threshold 0.21 is 0.4813165292649799\nF1 score at threshold 0.22 is 0.4856202466926567\nF1 score at threshold 0.23 is 0.49060937929695714\nF1 score at threshold 0.24 is 0.49540833865344003\nF1 score at threshold 0.25 is 0.500041372234266\nF1 score at threshold 0.26 is 0.5043236824058742\nF1 score at threshold 0.27 is 0.5084201814097259\nF1 score at threshold 0.28 is 0.512854655331353\nF1 score at threshold 0.29 is 0.5169202111145607\nF1 score at threshold 0.3 is 0.5204668176275911\nF1 score at threshold 0.31 is 0.5241143982833222\nF1 score at threshold 0.32 is 0.5278280944770023\nF1 score at threshold 0.33 is 0.5315730075416942\nF1 score at threshold 0.34 is 0.5348921253297676\nF1 score at threshold 0.35 is 0.5383115581552143\nF1 score at threshold 0.36 is 0.5416536431999117\nF1 score at threshold 0.37 is 0.5446724882733559\nF1 score at threshold 0.38 is 0.548072248607008\nF1 score at threshold 0.39 is 0.551010015277542\nF1 score at threshold 0.4 is 0.5541792720284643\nF1 score at threshold 0.41 is 0.5571795757087504\nF1 score at threshold 0.42 is 0.5600989410218752\nF1 score at threshold 0.43 is 0.5628835849975645\nF1 score at threshold 0.44 is 0.5661540878650406\nF1 score at threshold 0.45 is 0.569143038175194\nF1 score at threshold 0.46 is 0.5722669115151999\nF1 score at threshold 0.47 is 0.575087451248442\nF1 score at threshold 0.48 is 0.5780100893454082\nF1 score at threshold 0.49 is 0.580664924032021\nF1 score at threshold 0.5 is 0.5840693389331523\n","name":"stdout"}]},{"metadata":{"trusted":true},"cell_type":"code","source":"pred_glove_train_y = model.predict([train_X], batch_size=1024, verbose=1)","execution_count":23,"outputs":[{"output_type":"stream","text":"162532/162532 [==============================] - 3s 17us/step\n","name":"stdout"}]},{"metadata":{"trusted":true},"cell_type":"code","source":"pred_glove_test_y = model.predict([test_X], batch_size=1024, verbose=1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"del word_index, embeddings_index, all_embs, embedding_matrix, model, inp, x\nimport gc; gc.collect()\ntime.sleep(10)","execution_count":24,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"EMBEDDING_FILE = '../input/quora-insincere-questions-classification/embeddings/wiki-news-300d-1M/wiki-news-300d-1M.vec'\ndef get_coefs(word,*arr): return word, np.asarray(arr, dtype='float32')\nembeddings_index = dict(get_coefs(*o.split(\" \")) for o in open(EMBEDDING_FILE) if len(o)>100)\n\nall_embs = np.stack(embeddings_index.values())\nemb_mean,emb_std = all_embs.mean(), all_embs.std()\nembed_size = all_embs.shape[1]\n\nword_index = tokenizer.word_index\nnb_words = min(max_features, len(word_index))\nembedding_matrix = np.random.normal(emb_mean, emb_std, (nb_words, embed_size))\nfor word, i in word_index.items():\n    if i >= max_features: continue\n    embedding_vector = embeddings_index.get(word)\n    if embedding_vector is not None: embedding_matrix[i] = embedding_vector\n        \ninp = Input(shape=(maxlen,))\nx = Embedding(max_features, embed_size, weights=[embedding_matrix])(inp)\nx = Bidirectional(CuDNNGRU(64, return_sequences=True))(x)\nx = GlobalMaxPool1D()(x)\nx = Dense(16, activation=\"relu\")(x)\nx = Dropout(0.1)(x)\nx = Dense(1, activation=\"sigmoid\")(x)\nmodel = Model(inputs=inp, outputs=x)\nmodel.compile(loss='binary_crossentropy', optimizer='adam', metrics=['accuracy'])","execution_count":25,"outputs":[{"output_type":"stream","text":"/opt/conda/lib/python3.6/site-packages/ipykernel_launcher.py:5: FutureWarning: arrays to stack must be passed as a \"sequence\" type such as list or tuple. Support for non-sequence iterables such as generators is deprecated as of NumPy 1.16 and will raise an error in the future.\n  \"\"\"\n","name":"stderr"}]},{"metadata":{"trusted":true},"cell_type":"code","source":"model.fit(train_X, train_y, batch_size=512, epochs=2, validation_data=(val_X, val_y))","execution_count":26,"outputs":[{"output_type":"stream","text":"Train on 162532 samples, validate on 261225 samples\nEpoch 1/2\n162532/162532 [==============================] - 19s 119us/step - loss: 0.2992 - acc: 0.8768 - val_loss: 0.2403 - val_acc: 0.8975\nEpoch 2/2\n162532/162532 [==============================] - 17s 107us/step - loss: 0.2124 - acc: 0.9188 - val_loss: 0.2370 - val_acc: 0.9011\n","name":"stdout"},{"output_type":"execute_result","execution_count":26,"data":{"text/plain":"<keras.callbacks.History at 0x7fcbf2a41fd0>"},"metadata":{}}]},{"metadata":{"trusted":true},"cell_type":"code","source":"pred_fasttext_val_y = model.predict([val_X], batch_size=1024, verbose=1)\nfor thresh in np.arange(0.1, 0.501, 0.01):\n    thresh = np.round(thresh, 2)\n    print(\"F1 score at threshold {0} is {1}\".format(thresh, metrics.f1_score(val_y, (pred_fasttext_val_y>thresh).astype(int))))","execution_count":27,"outputs":[{"output_type":"stream","text":"261225/261225 [==============================] - 4s 17us/step\nF1 score at threshold 0.1 is 0.33945514779196556\nF1 score at threshold 0.11 is 0.3483631871525633\nF1 score at threshold 0.12 is 0.3568477879511527\nF1 score at threshold 0.13 is 0.36481005029784674\nF1 score at threshold 0.14 is 0.3721470110980824\nF1 score at threshold 0.15 is 0.37922857281130523\nF1 score at threshold 0.16 is 0.38584075160900544\nF1 score at threshold 0.17 is 0.39241348992024117\nF1 score at threshold 0.18 is 0.39897743976481115\nF1 score at threshold 0.19 is 0.40491722327624297\nF1 score at threshold 0.2 is 0.4109277778511698\nF1 score at threshold 0.21 is 0.4160823691866311\nF1 score at threshold 0.22 is 0.4218048182931809\nF1 score at threshold 0.23 is 0.4270137470325181\nF1 score at threshold 0.24 is 0.4320709579037201\nF1 score at threshold 0.25 is 0.43678356358483883\nF1 score at threshold 0.26 is 0.441641917387624\nF1 score at threshold 0.27 is 0.4462224287394543\nF1 score at threshold 0.28 is 0.45028346502952143\nF1 score at threshold 0.29 is 0.454877614068441\nF1 score at threshold 0.3 is 0.45919853460054355\nF1 score at threshold 0.31 is 0.4633457790729442\nF1 score at threshold 0.32 is 0.4673803024261571\nF1 score at threshold 0.33 is 0.4714615587273544\nF1 score at threshold 0.34 is 0.4757505050346869\nF1 score at threshold 0.35 is 0.47958376822595444\nF1 score at threshold 0.36 is 0.48347926120023005\nF1 score at threshold 0.37 is 0.4875016147784524\nF1 score at threshold 0.38 is 0.4910845258662842\nF1 score at threshold 0.39 is 0.49467848859206137\nF1 score at threshold 0.4 is 0.49804051810029887\nF1 score at threshold 0.41 is 0.5012987448259681\nF1 score at threshold 0.42 is 0.5043766264490182\nF1 score at threshold 0.43 is 0.5079868391892123\nF1 score at threshold 0.44 is 0.511595898424059\nF1 score at threshold 0.45 is 0.5149363836767284\nF1 score at threshold 0.46 is 0.51809970592354\nF1 score at threshold 0.47 is 0.5212435598842543\nF1 score at threshold 0.48 is 0.5246047571571001\nF1 score at threshold 0.49 is 0.5281086616719668\nF1 score at threshold 0.5 is 0.5317424338585961\n","name":"stdout"}]},{"metadata":{"trusted":true},"cell_type":"code","source":"pred_fasttext_train_y = model.predict([train_X], batch_size=1024, verbose=1)","execution_count":28,"outputs":[{"output_type":"stream","text":"162532/162532 [==============================] - 3s 16us/step\n","name":"stdout"}]},{"metadata":{"trusted":true},"cell_type":"code","source":"del word_index, embeddings_index, all_embs, embedding_matrix, model, inp, x\nimport gc; gc.collect()\ntime.sleep(10)","execution_count":29,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"EMBEDDING_FILE = '../input/quora-insincere-questions-classification/embeddings/paragram_300_sl999/paragram_300_sl999.txt'\ndef get_coefs(word,*arr): return word, np.asarray(arr, dtype='float32')\nembeddings_index = dict(get_coefs(*o.split(\" \")) for o in open(EMBEDDING_FILE, encoding=\"utf8\", errors='ignore'))\n\nall_embs = np.stack(embeddings_index.values())\nemb_mean,emb_std = all_embs.mean(), all_embs.std()\nembed_size = all_embs.shape[1]\n\nword_index = tokenizer.word_index\nnb_words = min(max_features, len(word_index))\nembedding_matrix = np.random.normal(emb_mean, emb_std, (nb_words, embed_size))\nfor word, i in word_index.items():\n    if i >= max_features: continue\n    embedding_vector = embeddings_index.get(word)\n    if embedding_vector is not None: embedding_matrix[i] = embedding_vector\n        \ninp = Input(shape=(maxlen,))\nx = Embedding(max_features, embed_size, weights=[embedding_matrix])(inp)\nx = Bidirectional(CuDNNGRU(64, return_sequences=True))(x)\nx = GlobalMaxPool1D()(x)\nx = Dense(16, activation=\"relu\")(x)\nx = Dropout(0.1)(x)\nx = Dense(1, activation=\"sigmoid\")(x)\nmodel = Model(inputs=inp, outputs=x)\nmodel.compile(loss='binary_crossentropy', optimizer='adam', metrics=['accuracy'])\nprint(model.summary())","execution_count":30,"outputs":[{"output_type":"stream","text":"/opt/conda/lib/python3.6/site-packages/ipykernel_launcher.py:5: FutureWarning: arrays to stack must be passed as a \"sequence\" type such as list or tuple. Support for non-sequence iterables such as generators is deprecated as of NumPy 1.16 and will raise an error in the future.\n  \"\"\"\n","name":"stderr"},{"output_type":"stream","text":"_________________________________________________________________\nLayer (type)                 Output Shape              Param #   \n=================================================================\ninput_3 (InputLayer)         (None, 100)               0         \n_________________________________________________________________\nembedding_3 (Embedding)      (None, 100, 300)          15000000  \n_________________________________________________________________\nbidirectional_3 (Bidirection (None, 100, 128)          140544    \n_________________________________________________________________\nglobal_max_pooling1d_3 (Glob (None, 128)               0         \n_________________________________________________________________\ndense_5 (Dense)              (None, 16)                2064      \n_________________________________________________________________\ndropout_3 (Dropout)          (None, 16)                0         \n_________________________________________________________________\ndense_6 (Dense)              (None, 1)                 17        \n=================================================================\nTotal params: 15,142,625\nTrainable params: 15,142,625\nNon-trainable params: 0\n_________________________________________________________________\nNone\n","name":"stdout"}]},{"metadata":{"trusted":true},"cell_type":"code","source":"model.fit(train_X, train_y, batch_size=512, epochs=2, validation_data=(val_X, val_y))","execution_count":31,"outputs":[{"output_type":"stream","text":"Train on 162532 samples, validate on 261225 samples\nEpoch 1/2\n162532/162532 [==============================] - 21s 127us/step - loss: 0.3029 - acc: 0.8762 - val_loss: 0.2188 - val_acc: 0.9074\nEpoch 2/2\n162532/162532 [==============================] - 18s 113us/step - loss: 0.2278 - acc: 0.9138 - val_loss: 0.2065 - val_acc: 0.9121\n","name":"stdout"},{"output_type":"execute_result","execution_count":31,"data":{"text/plain":"<keras.callbacks.History at 0x7fcbb8f2c828>"},"metadata":{}}]},{"metadata":{"trusted":true},"cell_type":"code","source":"pred_paragram_val_y = model.predict([val_X], batch_size=1024, verbose=1)\nfor thresh in np.arange(0.1, 0.501, 0.01):\n    thresh = np.round(thresh, 2)\n    print(\"F1 score at threshold {0} is {1}\".format(thresh, metrics.f1_score(val_y, (pred_paragram_val_y>thresh).astype(int))))\n","execution_count":32,"outputs":[{"output_type":"stream","text":"261225/261225 [==============================] - 5s 18us/step\nF1 score at threshold 0.1 is 0.36953866391055157\nF1 score at threshold 0.11 is 0.3788502638283486\nF1 score at threshold 0.12 is 0.38733231028166915\nF1 score at threshold 0.13 is 0.3957932083122149\nF1 score at threshold 0.14 is 0.4032339434706681\nF1 score at threshold 0.15 is 0.4102672385351369\nF1 score at threshold 0.16 is 0.4168884587466731\nF1 score at threshold 0.17 is 0.42356095425694906\nF1 score at threshold 0.18 is 0.4295402746122631\nF1 score at threshold 0.19 is 0.43535571644297666\nF1 score at threshold 0.2 is 0.44101389087784626\nF1 score at threshold 0.21 is 0.44646567381038865\nF1 score at threshold 0.22 is 0.451707935011563\nF1 score at threshold 0.23 is 0.4569338707269741\nF1 score at threshold 0.24 is 0.46155008537709474\nF1 score at threshold 0.25 is 0.4659176029962546\nF1 score at threshold 0.26 is 0.47077389609184017\nF1 score at threshold 0.27 is 0.47546933667083857\nF1 score at threshold 0.28 is 0.47940548659973115\nF1 score at threshold 0.29 is 0.48293804066779455\nF1 score at threshold 0.3 is 0.48687520154788777\nF1 score at threshold 0.31 is 0.4909022789099024\nF1 score at threshold 0.32 is 0.4947503538162789\nF1 score at threshold 0.33 is 0.49861191920871073\nF1 score at threshold 0.34 is 0.5020891014346841\nF1 score at threshold 0.35 is 0.5058713886300094\nF1 score at threshold 0.36 is 0.5093222948239883\nF1 score at threshold 0.37 is 0.5128789448242524\nF1 score at threshold 0.38 is 0.5167497472722836\nF1 score at threshold 0.39 is 0.5201505134336757\nF1 score at threshold 0.4 is 0.5239226813264763\nF1 score at threshold 0.41 is 0.5270103535218068\nF1 score at threshold 0.42 is 0.5302904564315353\nF1 score at threshold 0.43 is 0.5337604193207877\nF1 score at threshold 0.44 is 0.5372076801644701\nF1 score at threshold 0.45 is 0.5402605091770278\nF1 score at threshold 0.46 is 0.5438494988895317\nF1 score at threshold 0.47 is 0.5472717348340831\nF1 score at threshold 0.48 is 0.5502627435357502\nF1 score at threshold 0.49 is 0.5531613865361452\nF1 score at threshold 0.5 is 0.5569082945841132\n","name":"stdout"}]},{"metadata":{"trusted":true},"cell_type":"code","source":"pred_paragram_train_y = model.predict([train_X], batch_size=1024, verbose=1)","execution_count":33,"outputs":[{"output_type":"stream","text":"162532/162532 [==============================] - 3s 17us/step\n","name":"stdout"}]},{"metadata":{"trusted":true},"cell_type":"code","source":"del word_index, embeddings_index, all_embs, embedding_matrix, model, inp, x\nimport gc; gc.collect()\ntime.sleep(10)","execution_count":34,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"pred_val_y = 0.33*pred_glove_val_y + 0.33*pred_fasttext_val_y + 0.34*pred_paragram_val_y \nfor thresh in np.arange(0.1, 0.501, 0.01):\n    thresh = np.round(thresh, 2)\n    print(\"F1 score at threshold {0} is {1}\".format(thresh, metrics.f1_score(val_y, (pred_val_y>thresh).astype(int))))\n","execution_count":35,"outputs":[{"output_type":"stream","text":"F1 score at threshold 0.1 is 0.3621718855623919\nF1 score at threshold 0.11 is 0.3719756042893039\nF1 score at threshold 0.12 is 0.38080825201578733\nF1 score at threshold 0.13 is 0.3893223513650967\nF1 score at threshold 0.14 is 0.397594793794162\nF1 score at threshold 0.15 is 0.4049659529296016\nF1 score at threshold 0.16 is 0.4120885649780172\nF1 score at threshold 0.17 is 0.4189543626508131\nF1 score at threshold 0.18 is 0.4254451951746588\nF1 score at threshold 0.19 is 0.4320963614109269\nF1 score at threshold 0.2 is 0.438169232942506\nF1 score at threshold 0.21 is 0.44423093469469704\nF1 score at threshold 0.22 is 0.44971189104243064\nF1 score at threshold 0.23 is 0.4551400351379682\nF1 score at threshold 0.24 is 0.4603207852055779\nF1 score at threshold 0.25 is 0.4656346185920277\nF1 score at threshold 0.26 is 0.47045458034603993\nF1 score at threshold 0.27 is 0.4753660128687327\nF1 score at threshold 0.28 is 0.48030839430414607\nF1 score at threshold 0.29 is 0.48529528844163494\nF1 score at threshold 0.3 is 0.4898872785829308\nF1 score at threshold 0.31 is 0.4942291100294649\nF1 score at threshold 0.32 is 0.4990117275003294\nF1 score at threshold 0.33 is 0.5029368209121616\nF1 score at threshold 0.34 is 0.5069844845265511\nF1 score at threshold 0.35 is 0.5115433619298394\nF1 score at threshold 0.36 is 0.5158377255480969\nF1 score at threshold 0.37 is 0.5201249132546842\nF1 score at threshold 0.38 is 0.524388960970649\nF1 score at threshold 0.39 is 0.5283493288947125\nF1 score at threshold 0.4 is 0.5320648091848779\nF1 score at threshold 0.41 is 0.5360005777424712\nF1 score at threshold 0.42 is 0.5398381688292754\nF1 score at threshold 0.43 is 0.5432380251907696\nF1 score at threshold 0.44 is 0.5468187051963005\nF1 score at threshold 0.45 is 0.5503524295140972\nF1 score at threshold 0.46 is 0.5540906081528421\nF1 score at threshold 0.47 is 0.5576750491684329\nF1 score at threshold 0.48 is 0.5613243909959913\nF1 score at threshold 0.49 is 0.564434593249339\nF1 score at threshold 0.5 is 0.5684508037449214\n","name":"stdout"}]},{"metadata":{"trusted":true},"cell_type":"code","source":"pred_train_y = 0.33*pred_glove_train_y + 0.33*pred_fasttext_train_y + 0.34*pred_paragram_train_y \nfor thresh in np.arange(0.1, 0.501, 0.01):\n    thresh = np.round(thresh, 2)\n    print(\"F1 score at threshold {0} is {1}\".format(thresh, metrics.f1_score(val_y, (pred_val_y>thresh).astype(int))))\n","execution_count":36,"outputs":[{"output_type":"stream","text":"F1 score at threshold 0.1 is 0.3621718855623919\nF1 score at threshold 0.11 is 0.3719756042893039\nF1 score at threshold 0.12 is 0.38080825201578733\nF1 score at threshold 0.13 is 0.3893223513650967\nF1 score at threshold 0.14 is 0.397594793794162\nF1 score at threshold 0.15 is 0.4049659529296016\nF1 score at threshold 0.16 is 0.4120885649780172\nF1 score at threshold 0.17 is 0.4189543626508131\nF1 score at threshold 0.18 is 0.4254451951746588\nF1 score at threshold 0.19 is 0.4320963614109269\nF1 score at threshold 0.2 is 0.438169232942506\nF1 score at threshold 0.21 is 0.44423093469469704\nF1 score at threshold 0.22 is 0.44971189104243064\nF1 score at threshold 0.23 is 0.4551400351379682\nF1 score at threshold 0.24 is 0.4603207852055779\nF1 score at threshold 0.25 is 0.4656346185920277\nF1 score at threshold 0.26 is 0.47045458034603993\nF1 score at threshold 0.27 is 0.4753660128687327\nF1 score at threshold 0.28 is 0.48030839430414607\nF1 score at threshold 0.29 is 0.48529528844163494\nF1 score at threshold 0.3 is 0.4898872785829308\nF1 score at threshold 0.31 is 0.4942291100294649\nF1 score at threshold 0.32 is 0.4990117275003294\nF1 score at threshold 0.33 is 0.5029368209121616\nF1 score at threshold 0.34 is 0.5069844845265511\nF1 score at threshold 0.35 is 0.5115433619298394\nF1 score at threshold 0.36 is 0.5158377255480969\nF1 score at threshold 0.37 is 0.5201249132546842\nF1 score at threshold 0.38 is 0.524388960970649\nF1 score at threshold 0.39 is 0.5283493288947125\nF1 score at threshold 0.4 is 0.5320648091848779\nF1 score at threshold 0.41 is 0.5360005777424712\nF1 score at threshold 0.42 is 0.5398381688292754\nF1 score at threshold 0.43 is 0.5432380251907696\nF1 score at threshold 0.44 is 0.5468187051963005\nF1 score at threshold 0.45 is 0.5503524295140972\nF1 score at threshold 0.46 is 0.5540906081528421\nF1 score at threshold 0.47 is 0.5576750491684329\nF1 score at threshold 0.48 is 0.5613243909959913\nF1 score at threshold 0.49 is 0.564434593249339\nF1 score at threshold 0.5 is 0.5684508037449214\n","name":"stdout"}]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":61,"outputs":[{"output_type":"execute_result","execution_count":61,"data":{"text/plain":"numpy.ndarray"},"metadata":{}}]},{"metadata":{"trusted":true},"cell_type":"code","source":"#pred_train_df = pd.DataFrame({\"prediction\":pred_train_y}, index=[0])\npd.DataFrame(pred_train_y).to_csv(\"pred_train.csv\", header=None, index=None)","execution_count":62,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#pred_train_df = pd.DataFrame({\"prediction\":pred_train_y}, index=[0])\npd.DataFrame(pred_val_y).to_csv(\"pred_val.csv\", header=None, index=None)","execution_count":63,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.4","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}