{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"scrolled":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nimport os\nprint(os.listdir(\"../input\"))\n\n# Any results you write to the current directory are saved as output.","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"train_df = pd.read_csv('../input/train.csv')\nX_train = train_df[\"question_text\"].fillna(\"dieter\").values\ntest_df = pd.read_csv('../input/test.csv')\nX_test = test_df[\"question_text\"].fillna(\"dieter\").values\ny = train_df[\"target\"]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d565c03e9b7b2f133e4e8f4eb0ed687c355f1fc4"},"cell_type":"code","source":"from keras.models import Model\nfrom keras.layers import Input, Dense, Embedding, concatenate\nfrom keras.layers import CuDNNGRU, Bidirectional, GlobalAveragePooling1D, GlobalMaxPooling1D, Conv1D\nfrom keras.layers import Add, BatchNormalization, Activation, CuDNNLSTM, Dropout\nfrom keras.layers import *\nfrom keras.models import *\nfrom keras.preprocessing import text, sequence\nfrom keras.callbacks import EarlyStopping, ModelCheckpoint, ReduceLROnPlateau\nimport gc\nfrom sklearn import metrics\nfrom keras.optimizers import *","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ca0edbb51d3a75ca4d667c5eca1dcdfb916a7fa7"},"cell_type":"code","source":"maxlen = 70\nmax_features = 50000\nembed_size = 300\n\ntokenizer = text.Tokenizer(num_words=max_features)\ntokenizer.fit_on_texts(list(X_train) + list(X_test))\n\nX_train = tokenizer.texts_to_sequences(X_train)\nX_test = tokenizer.texts_to_sequences(X_test)\n\nx_train = sequence.pad_sequences(X_train, maxlen=maxlen)\nx_test = sequence.pad_sequences(X_test, maxlen=maxlen)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"3f5c8eee53cf45ccfd1ce56f4011f1df00186a5e"},"cell_type":"code","source":"def attention_3d_block(inputs):\n    # inputs.shape = (batch_size, time_steps, input_dim)\n    TIME_STEPS = inputs.shape[1].value\n    SINGLE_ATTENTION_VECTOR = False\n    \n    input_dim = int(inputs.shape[2])\n    a = Permute((2, 1))(inputs)\n    a = Reshape((input_dim, TIME_STEPS))(a) # this line is not useful. It's just to know which dimension is what.\n    a = Dense(TIME_STEPS, activation='softmax')(a)\n    if SINGLE_ATTENTION_VECTOR:\n        a = Lambda(lambda x: K.mean(x, axis=1))(a)\n        a = RepeatVector(input_dim)(a)\n    a_probs = Permute((2, 1))(a)\n    output_attention_mul = Multiply()([inputs, a_probs])\n    return output_attention_mul","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"43ba65d695360fe604e235ead98e2bc38ecaeb1b"},"cell_type":"code","source":"from keras import backend as K\nfrom keras.engine.topology import Layer, InputSpec\nfrom keras import initializers\n\nclass AttLayer(Layer):\n    def __init__(self, attention_dim):\n        self.init = initializers.get('normal')\n        self.supports_masking = True\n        self.attention_dim = attention_dim\n        super(AttLayer, self).__init__()\n\n    def build(self, input_shape):\n        assert len(input_shape) == 3\n        self.W = K.variable(self.init((input_shape[-1], self.attention_dim)))\n        self.b = K.variable(self.init((self.attention_dim, )))\n        self.u = K.variable(self.init((self.attention_dim, 1)))\n        self.trainable_weights = [self.W, self.b, self.u]\n        super(AttLayer, self).build(input_shape)\n\n    def compute_mask(self, inputs, mask=None):\n        return mask\n\n    def call(self, x, mask=None):\n        # size of x :[batch_size, sel_len, attention_dim]\n        # size of u :[batch_size, attention_dim]\n        # uit = tanh(xW+b)\n        uit = K.tanh(K.bias_add(K.dot(x, self.W), self.b))\n        ait = K.dot(uit, self.u)\n        ait = K.squeeze(ait, -1)\n\n        ait = K.exp(ait)\n\n        if mask is not None:\n            # Cast the mask to floatX to avoid float64 upcasting in theano\n            ait *= K.cast(mask, K.floatx())\n        ait /= K.cast(K.sum(ait, axis=1, keepdims=True) + K.epsilon(), K.floatx())\n        ait = K.expand_dims(ait)\n        weighted_input = x * ait\n        output = K.sum(weighted_input, axis=1)\n\n        return output\n\n    def compute_output_shape(self, input_shape):\n        return (input_shape[0], input_shape[-1])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"1ae3acbc750037108d8d0eec7850d5209a1ecbc8"},"cell_type":"code","source":"def noise_measurement(train, noise_level):\n    noised_train = train.copy()\n    to_transform = np.random.random(train.shape) < noise_level\n    transform_y = np.random.randint(0, max_features, size=train.shape)\n    noised_train[to_transform] = transform_y[to_transform]\n    return noised_train","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"def48377e4113ab78d5b5bafc323f36d3b9232ba"},"cell_type":"code","source":"EMBEDDING_FILE = '../input/embeddings/glove.840B.300d/glove.840B.300d.txt'\ndef get_coefs(word,*arr): return word, np.asarray(arr, dtype='float32')\nembeddings_index = dict(get_coefs(*o.split(\" \")) for o in open(EMBEDDING_FILE))\n\nall_embs = np.stack(embeddings_index.values())\nemb_mean,emb_std = all_embs.mean(), all_embs.std()\nembed_size = all_embs.shape[1]\n\nword_index = tokenizer.word_index\nnb_words = min(max_features, len(word_index))\nembedding_matrix_1 = np.random.normal(emb_mean, emb_std, (nb_words, embed_size))\nfor word, i in word_index.items():\n    if i >= max_features: continue\n    embedding_vector = embeddings_index.get(word)\n    if embedding_vector is not None: embedding_matrix_1[i] = embedding_vector\n\ndel embeddings_index; gc.collect() ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"29c446e36dc3e0cfeb86b35828664f739d3b3eae"},"cell_type":"code","source":"EMBEDDING_FILE = '../input/embeddings/wiki-news-300d-1M/wiki-news-300d-1M.vec'\ndef get_coefs(word,*arr): return word, np.asarray(arr, dtype='float32')\nembeddings_index = dict(get_coefs(*o.split(\" \")) for o in open(EMBEDDING_FILE) if len(o)>100)\n\nall_embs = np.stack(embeddings_index.values())\nemb_mean,emb_std = all_embs.mean(), all_embs.std()\nembed_size = all_embs.shape[1]\n\nword_index = tokenizer.word_index\nnb_words = min(max_features, len(word_index))\nembedding_matrix_2 = np.random.normal(emb_mean, emb_std, (nb_words, embed_size))\nfor word, i in word_index.items():\n    if i >= max_features: continue\n    embedding_vector = embeddings_index.get(word)\n    if embedding_vector is not None: embedding_matrix_2[i] = embedding_vector\ndel embeddings_index; gc.collect()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"1cf537631d802e0d8434d75e8a3f5f39d1f73477"},"cell_type":"code","source":"EMBEDDING_FILE = '../input/embeddings/paragram_300_sl999/paragram_300_sl999.txt'\ndef get_coefs(word,*arr): return word, np.asarray(arr, dtype='float32')\nembeddings_index = dict(get_coefs(*o.split(\" \")) for o in open(EMBEDDING_FILE, encoding=\"utf8\", errors='ignore') if len(o)>100)\n\nall_embs = np.stack(embeddings_index.values())\nemb_mean,emb_std = all_embs.mean(), all_embs.std()\nembed_size = all_embs.shape[1]\n\nword_index = tokenizer.word_index\nnb_words = min(max_features, len(word_index))\nembedding_matrix_3 = np.random.normal(emb_mean, emb_std, (nb_words, embed_size))\nfor word, i in word_index.items():\n    if i >= max_features: continue\n    embedding_vector = embeddings_index.get(word)\n    if embedding_vector is not None: embedding_matrix_3[i] = embedding_vector\n        \ndel embeddings_index; gc.collect()   ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f741d8d29bd523d6dae9f077ea602c00e066b89c"},"cell_type":"code","source":"# # https://www.kaggle.com/strideradu/word2vec-and-gensim-go-go-go\n# from gensim.models import KeyedVectors\n\n# EMBEDDING_FILE = '../input/embeddings/GoogleNews-vectors-negative300/GoogleNews-vectors-negative300.bin'\n# embeddings_index = KeyedVectors.load_word2vec_format(EMBEDDING_FILE, binary=True)\n\n# word_index = tokenizer.word_index\n# nb_words = min(max_features, len(word_index))\n# embedding_matrix_4 = (np.random.rand(nb_words, embed_size) - 0.5) / 5.0\n# for word, i in word_index.items():\n#     if i >= max_features: continue\n#     if word in embeddings_index:\n#         embedding_vector = embeddings_index.get_vector(word)\n#         embedding_matrix_4[i] = embedding_vector\n        \n# del embeddings_index; gc.collect()        ","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"5693a3faaabc51c49434b7057ec68fe3f6729756"},"cell_type":"markdown","source":"# Concatenating the embeddings"},{"metadata":{"trusted":true,"_uuid":"7d891d3452e22c4fa9f7e2781527618041a38f83"},"cell_type":"code","source":"embedding_matrix = np.mean([embedding_matrix_1, embedding_matrix_2, embedding_matrix_3], axis=0)  \ndel embedding_matrix_1, embedding_matrix_2, embedding_matrix_3\ngc.collect()\nnp.shape(embedding_matrix)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"2f14547280cb1ee50b34dc2d884d878520177073"},"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nX_tra, X_val, y_tra, y_val = train_test_split(x_train, y, test_size = 0.1, random_state=42)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"2422154c8ee1bba339e83a7a63f18a3fc78efbea"},"cell_type":"markdown","source":"# MODEL 1: Conv2D"},{"metadata":{"trusted":true,"_uuid":"0239f07651f3ec14ee357206352ba39607ff3a68"},"cell_type":"code","source":"def model1():\n    inp = Input(shape=(maxlen, ))\n    embed = Embedding(max_features, embed_size * 1, weights=[embedding_matrix], trainable=False)(inp)\n    embed = SpatialDropout1D(0.1)(embed)\n    \n    #x = Reshape((maxlen, embed_size * 3, 1))(embed)\n    \n    filter_sizes = [1,2,3,5]\n    num_filters = 64\n    \n    conv_0 = Conv1D(num_filters, filter_sizes[0], padding='valid', kernel_initializer='normal', activation='relu')(embed)\n    conv_1 = Conv1D(num_filters, filter_sizes[1], padding='valid', kernel_initializer='normal', activation='relu')(embed)\n    conv_2 = Conv1D(num_filters, filter_sizes[2], padding='valid', kernel_initializer='normal', activation='relu')(embed)\n    conv_3 = Conv1D(num_filters, filter_sizes[3], padding='valid', kernel_initializer='normal', activation='relu')(embed)\n\n    maxpool_0 = MaxPool1D(pool_size=(maxlen - filter_sizes[0] + 1), strides=(1), padding='valid')(conv_0)\n    maxpool_1 = MaxPool1D(pool_size=(maxlen - filter_sizes[1] + 1), strides=(1), padding='valid')(conv_1)\n    maxpool_2 = MaxPool1D(pool_size=(maxlen - filter_sizes[2] + 1), strides=(1), padding='valid')(conv_2)\n    maxpool_3 = MaxPool1D(pool_size=(maxlen - filter_sizes[3] + 1), strides=(1), padding='valid')(conv_3)\n\n    concatenated_tensor = Concatenate(axis=1)([maxpool_0, maxpool_1, maxpool_2, maxpool_3])\n    \n    #gmp = GlobalMaxPooling1D()(concatenated_tensor)\n    #gap = GlobalAveragePooling1D()(concatenated_tensor)\n    \n    #conc = Concatenate(axis=1)([gmp, gap])\n    \n    flatten = Flatten()(concatenated_tensor)\n    \n    x = flatten\n    x = Dropout(0.5)(x)\n    x = Dense(128, activation='relu')(x)\n    outp = Dense(1, activation=\"sigmoid\")(x)\n    model = Model(inputs=inp, outputs=outp)\n    model.compile(loss='binary_crossentropy',\n                  optimizer=Adam(lr=1e-3, decay=1e-6),\n                  metrics=['accuracy'])    \n\n    return model","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"5e7a9314bb3a1214c59279424840936890f4d146","scrolled":true},"cell_type":"code","source":"MODEL1 = model1()\nMODEL1.summary()\n\nbatch_size = 2048\nepochs = 4\n\nhist = MODEL1.fit(X_tra, y_tra, batch_size=batch_size, epochs=epochs, validation_data=(X_val, y_val), verbose=True)\nMODEL1.save('./model1.h5')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"fef529e912bdc9580d4044221e11547a2d8d0bfa","scrolled":true},"cell_type":"code","source":"pred_val_y_1 = MODEL1.predict([X_val], batch_size=1024, verbose=1)\nthresholds = []\nfor thresh in np.arange(0.1, 0.501, 0.01):\n    thresh = np.round(thresh, 2)\n    res = metrics.f1_score(y_val, (pred_val_y_1 > thresh).astype(int))\n    thresholds.append([thresh, res])\n    print(\"F1 score at threshold {0} is {1}\".format(thresh, res))\n    \nthresholds.sort(key=lambda x: x[1], reverse=True)\nbest_thresh_1 = thresholds[0][0]\nprint(\"Best threshold: \", best_thresh_1)\n\ny_pred_1 = MODEL1.predict(x_test, batch_size=1024, verbose=True)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"48b25baac8abdcf0ec2423f97b2b282f93c476de"},"cell_type":"markdown","source":"# MODEL 2: LSTM"},{"metadata":{"trusted":true,"_uuid":"37e7f8f9e9c1f21011e91b84f56b545f0934835e"},"cell_type":"code","source":"def model2():\n    inp = Input(shape=(maxlen, ))\n    embed = Embedding(max_features, embed_size * 1, weights=[embedding_matrix], trainable=False)(inp)\n    x = embed\n    x = SpatialDropout1D(0.1)(x)\n    \n    x = Bidirectional(CuDNNLSTM(64, return_sequences=True))(x)\n    x = attention_3d_block(x)\n    x = Bidirectional(CuDNNLSTM(64, return_sequences=True))(x)\n    x = AttLayer(maxlen)(x)\n    x = Dropout(0.3)(x)\n    x = Dense(128, activation='relu')(x)\n    outp = Dense(1, activation=\"sigmoid\")(x)\n    model = Model(inputs=inp, outputs=outp)\n    model.compile(loss='binary_crossentropy',\n                  optimizer=Adam(lr=1e-3, decay=1e-6),\n                  metrics=['accuracy'])    \n\n    return model","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"scrolled":true,"_uuid":"2d5cb289d080483143edc805fe68de4ad8705b83"},"cell_type":"code","source":"MODEL2 = model2()\nMODEL2.summary()\n\nbatch_size = 2048\nepochs = 5\n\nhist = MODEL2.fit(X_tra, y_tra, batch_size=batch_size, epochs=epochs, validation_data=(X_val, y_val), verbose=True)\nMODEL2.save('./model2.h5')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a4ec75afa353a89fa6b785e6173e8e5584af6f4b","scrolled":true},"cell_type":"code","source":"pred_val_y_2 = MODEL2.predict([X_val], batch_size=1024, verbose=1)\nthresholds = []\nfor thresh in np.arange(0.1, 0.501, 0.01):\n    thresh = np.round(thresh, 2)\n    res = metrics.f1_score(y_val, (pred_val_y_2 > thresh).astype(int))\n    thresholds.append([thresh, res])\n    print(\"F1 score at threshold {0} is {1}\".format(thresh, res))\n    \nthresholds.sort(key=lambda x: x[1], reverse=True)\nbest_thresh_2 = thresholds[0][0]\nprint(\"Best threshold: \", best_thresh_2)\n\ny_pred_2 = MODEL2.predict(x_test, batch_size=1024, verbose=True)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d9c997b453417c3284558399df57a6ac98eedf83"},"cell_type":"markdown","source":"# MODEL 3: Conv1D"},{"metadata":{"trusted":true,"_uuid":"06ef150ff0f7fcc7c47d023f97f31b4580724068"},"cell_type":"code","source":"def model3():\n    filters = 64\n    \n    inp = Input(shape=(maxlen, ))\n    embed = Embedding(max_features, embed_size * 1, weights=[embedding_matrix], trainable=False)(inp)\n    x = embed\n    x = SpatialDropout1D(0.1)(x)\n    \n    x = Conv1D(filters, 1, activation='relu', padding='valid', kernel_initializer='normal')(x)\n    x = Dropout(0.1)(x)\n    \n    x = Conv1D(filters, 2, activation='relu', padding='valid', kernel_initializer='normal')(x)\n    x = Dropout(0.1)(x)\n    \n    x = Conv1D(filters, 3, activation='relu', padding='valid', kernel_initializer='normal')(x)\n    x = Dropout(0.1)(x)\n    \n    x = Conv1D(filters, 5, activation='relu', padding='valid', kernel_initializer='normal')(x)\n    x = Dropout(0.1)(x)\n    \n    x = Flatten()(x)\n    #x = GlobalAveragePooling1D()(x)\n    \n    x = Dropout(0.3)(x)\n    x = Dense(128, activation='relu')(x)\n    outp = Dense(1, activation=\"sigmoid\")(x)\n    model = Model(inputs=inp, outputs=outp)\n    model.compile(loss='binary_crossentropy',\n                  optimizer=Adam(lr=1e-3, decay=1e-6),\n                  metrics=['accuracy'])    \n\n    return model","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"scrolled":true,"_uuid":"4c3c4ce420cbb1161e88e9d64050a0b9f836042a"},"cell_type":"code","source":"MODEL3 = model3()\nMODEL3.summary()\n\nbatch_size = 2048\nepochs = 10\n\nhist = MODEL3.fit(X_tra, y_tra, batch_size=batch_size, epochs=epochs, validation_data=(X_val, y_val), verbose=True)\nMODEL3.save('./model3.h5')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c6ca1e0f7c17c3424f04b07212b0662ae49d7309","scrolled":true},"cell_type":"code","source":"pred_val_y_3 = MODEL3.predict([X_val], batch_size=1024, verbose=1)\nthresholds = []\nfor thresh in np.arange(0.1, 0.501, 0.01):\n    thresh = np.round(thresh, 2)\n    res = metrics.f1_score(y_val, (pred_val_y_3 > thresh).astype(int))\n    thresholds.append([thresh, res])\n    print(\"F1 score at threshold {0} is {1}\".format(thresh, res))\n    \nthresholds.sort(key=lambda x: x[1], reverse=True)\nbest_thresh_3 = thresholds[0][0]\nprint(\"Best threshold: \", best_thresh_3)\n\ny_pred_3 = MODEL3.predict(x_test, batch_size=1024, verbose=True)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"305c64b3e7c6002ce7a9ac63613f8a1b7aa2901b"},"cell_type":"markdown","source":"# MODEL 4: GRU"},{"metadata":{"trusted":true,"_uuid":"aa0e64fcd54a87d6ce152bf45e5caa902aaff035"},"cell_type":"code","source":"def model4():\n    inp = Input(shape=(maxlen, ))\n    embed = Embedding(max_features, embed_size * 1, weights=[embedding_matrix], trainable=False)(inp)\n    x = embed\n    x = SpatialDropout1D(0.1)(x)\n    \n    x = Bidirectional(CuDNNGRU(64, return_sequences=True))(x)\n    x = attention_3d_block(x)\n    x = Bidirectional(CuDNNGRU(64, return_sequences=True))(x)\n    x = AttLayer(64)(x)\n    \n    x = Dropout(0.3)(x)\n    x = Dense(128, activation='relu')(x)\n    outp = Dense(1, activation=\"sigmoid\")(x)\n    model = Model(inputs=inp, outputs=outp)\n    model.compile(loss='binary_crossentropy',\n                  optimizer=Adam(lr=1e-3, decay=1e-6),\n                  metrics=['accuracy'])    \n\n    return model","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"scrolled":true,"_uuid":"3ff5e82c52e11d83fe4d70e5d0fb0a8316125fc4"},"cell_type":"code","source":"MODEL4 = model4()\nMODEL4.summary()\n\nbatch_size = 2048\nepochs = 5\n\nhist = MODEL4.fit(X_tra, y_tra, batch_size=batch_size, epochs=epochs, validation_data=(X_val, y_val), verbose=True)\nMODEL4.save('./model4.h5')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"def9f127ac6118c7d14f094512069e1c3d26cff0","scrolled":true},"cell_type":"code","source":"pred_val_y_4 = MODEL4.predict([X_val], batch_size=1024, verbose=1)\nthresholds = []\nfor thresh in np.arange(0.1, 0.501, 0.01):\n    thresh = np.round(thresh, 2)\n    res = metrics.f1_score(y_val, (pred_val_y_4 > thresh).astype(int))\n    thresholds.append([thresh, res])\n    print(\"F1 score at threshold {0} is {1}\".format(thresh, res))\n    \nthresholds.sort(key=lambda x: x[1], reverse=True)\nbest_thresh_4 = thresholds[0][0]\nprint(\"Best threshold: \", best_thresh_4)\n\ny_pred_4 = MODEL4.predict(x_test, batch_size=1024, verbose=True)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"92c1f06de657421e831c59865616cf7774f1607a"},"cell_type":"markdown","source":"# Concat Result & Best Threshold"},{"metadata":{"trusted":true,"_uuid":"9dd8810a5348d5a26c63b2aa1b00642d09515120","scrolled":true},"cell_type":"code","source":"pred_val_y = (2.5*pred_val_y_1 + 2.5*pred_val_y_2 + 2.5*pred_val_y_3 + 2.5*pred_val_y_4)/10\n\nthresholds = []\nfor thresh in np.arange(0.1, 0.501, 0.01):\n    thresh = np.round(thresh, 2)\n    res = metrics.f1_score(y_val, (pred_val_y > thresh).astype(int))\n    thresholds.append([thresh, res])\n    print(\"F1 score at threshold {0} is {1}\".format(thresh, res))\n    \nthresholds.sort(key=lambda x: x[1], reverse=True)\nbest_thresh = thresholds[0][0]\nprint(\"Best threshold: \", best_thresh)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"37901249694b0c14f9ec5a4b6586f52686c92e2f"},"cell_type":"markdown","source":"# Submission File"},{"metadata":{"trusted":true,"_uuid":"404a671dca09cebb0a014796063bd59d82f98714"},"cell_type":"code","source":"y_pred = (2.5*y_pred_1 + 2.5*y_pred_2 + 2.5*y_pred_3 + 2.5*y_pred_4)/10\ny_te = (y_pred[:,0] > best_thresh).astype(np.int)\n\nsubmit_df = pd.DataFrame({\"qid\": test_df[\"qid\"], \"prediction\": y_te})\nsubmit_df.to_csv(\"submission.csv\", index=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d3c861699e0ae612641cc3eb7d288500c2168cd1"},"cell_type":"code","source":"from IPython.display import HTML\nimport base64  \nimport pandas as pd  \n\ndef create_download_link( df, title = \"Download CSV file\", filename = \"data.csv\"):  \n    csv = df.to_csv(index =False)\n    b64 = base64.b64encode(csv.encode())\n    payload = b64.decode()\n    html = '<a download=\"{filename}\" href=\"data:text/csv;base64,{payload}\" target=\"_blank\">{title}</a>'\n    html = html.format(payload=payload,title=title,filename=filename)\n    return HTML(html)\n\ncreate_download_link(submit_df)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"48313b5e086fd1dd08d43e51190db353a689cbbc"},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}