{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"#import Libraries\nimport os\nimport time\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nfrom tqdm import tqdm\nimport math\nfrom sklearn.model_selection import train_test_split\nfrom sklearn import metrics\nfrom keras.preprocessing import text, sequence\nfrom keras.preprocessing.text import Tokenizer\nfrom keras.preprocessing.sequence import pad_sequences\nfrom keras.layers import Dense, Input, CuDNNLSTM, Embedding, Dropout, Activation, CuDNNGRU, Conv1D,LSTM,GRU\nfrom keras.layers import GlobalAveragePooling1D, GlobalMaxPooling1D, concatenate, SpatialDropout1D\nfrom keras.layers import Bidirectional, GlobalMaxPool1D\nfrom keras.optimizers import Adam, Adadelta\nfrom keras.initializers import *\nfrom keras.models import Model\nfrom keras import backend as K\nfrom keras.engine.topology import Layer\nfrom keras import initializers, regularizers, constraints, optimizers, layers\nfrom keras.optimizers import Adam, RMSprop\nfrom keras.callbacks import EarlyStopping, ModelCheckpoint, LearningRateScheduler\nfrom keras.layers import GRU, BatchNormalization, Conv1D, MaxPooling1D\nimport logging\nfrom sklearn.metrics import roc_auc_score\nfrom keras.callbacks import Callback\nimport gensim.models.keyedvectors as word2vec","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"\ntrain = pd.read_csv('../input/train.csv')\ntest = pd.read_csv('../input/test.csv')\nsubmission = pd.read_csv('../input/sample_submission.csv')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"45439354936e9af75bd8f2e13f0a631cbeb3932d"},"cell_type":"code","source":"# PREPROCESSING PART\nfill = {\"ain't\": \"is not\", \"aren't\": \"are not\",\"can't\": \"cannot\", \n       \"can't've\": \"cannot have\", \"'cause\": \"because\", \"could've\": \"could have\", \n       \"couldn't\": \"could not\", \"couldn't've\": \"could not have\",\"didn't\": \"did not\", \n       \"doesn't\": \"does not\", \"don't\": \"do not\", \"hadn't\": \"had not\", \n       \"hadn't've\": \"had not have\", \"hasn't\": \"has not\", \"haven't\": \"have not\", \n       \"he'd\": \"he would\", \"he'd've\": \"he would have\", \"he'll\": \"he will\", \n       \"he'll've\": \"he he will have\", \"he's\": \"he is\", \"how'd\": \"how did\", \n       \"how'd'y\": \"how do you\", \"how'll\": \"how will\", \"how's\": \"how is\", \n       \"I'd\": \"I would\", \"I'd've\": \"I would have\", \"I'll\": \"I will\", \n       \"I'll've\": \"I will have\",\"I'm\": \"I am\", \"I've\": \"I have\", \n       \"i'd\": \"i would\", \"i'd've\": \"i would have\", \"i'll\": \"i will\", \n       \"i'll've\": \"i will have\",\"i'm\": \"i am\", \"i've\": \"i have\", \n       \"isn't\": \"is not\", \"it'd\": \"it would\", \"it'd've\": \"it would have\", \n       \"it'll\": \"it will\", \"it'll've\": \"it will have\",\"it's\": \"it is\", \n       \"let's\": \"let us\", \"ma'am\": \"madam\", \"mayn't\": \"may not\", \n       \"might've\": \"might have\",\"mightn't\": \"might not\",\"mightn't've\": \"might not have\", \n       \"must've\": \"must have\", \"mustn't\": \"must not\", \"mustn't've\": \"must not have\", \n       \"needn't\": \"need not\", \"needn't've\": \"need not have\",\"o'clock\": \"of the clock\", \n       \"oughtn't\": \"ought not\", \"oughtn't've\": \"ought not have\", \"shan't\": \"shall not\",\n       \"sha'n't\": \"shall not\", \"shan't've\": \"shall not have\", \"she'd\": \"she would\", \n       \"she'd've\": \"she would have\", \"she'll\": \"she will\", \"she'll've\": \"she will have\", \n       \"she's\": \"she is\", \"should've\": \"should have\", \"shouldn't\": \"should not\", \n       \"shouldn't've\": \"should not have\", \"so've\": \"so have\",\"so's\": \"so as\", \n       \"this's\": \"this is\",\n       \"that'd\": \"that would\", \"that'd've\": \"that would have\",\"that's\": \"that is\", \n       \"there'd\": \"there would\", \"there'd've\": \"there would have\",\"there's\": \"there is\", \n       \"they'd\": \"they would\", \"they'd've\": \"they would have\", \"they'll\": \"they will\", \n       \"they'll've\": \"they will have\", \"they're\": \"they are\", \"they've\": \"they have\", \n       \"to've\": \"to have\", \"wasn't\": \"was not\", \"we'd\": \"we would\", \n       \"we'd've\": \"we would have\", \"we'll\": \"we will\", \"we'll've\": \"we will have\", \n       \"we're\": \"we are\", \"we've\": \"we have\", \"weren't\": \"were not\", \n       \"what'll\": \"what will\", \"what'll've\": \"what will have\", \"what're\": \"what are\", \n       \"what's\": \"what is\", \"what've\": \"what have\", \"when's\": \"when is\", \n       \"when've\": \"when have\", \"where'd\": \"where did\", \"where's\": \"where is\", \n       \"where've\": \"where have\", \"who'll\": \"who will\", \"who'll've\": \"who will have\", \n       \"who's\": \"who is\", \"who've\": \"who have\", \"why's\": \"why is\", \n       \"why've\": \"why have\", \"will've\": \"will have\", \"won't\": \"will not\", \n       \"won't've\": \"will not have\", \"would've\": \"would have\", \"wouldn't\": \"would not\", \n       \"wouldn't've\": \"would not have\", \"y'all\": \"you all\", \"y'all'd\": \"you all would\",\n       \"y'all'd've\": \"you all would have\",\"y'all're\": \"you all are\",\"y'all've\": \"you all have\",\n       \"you'd\": \"you would\", \"you'd've\": \"you would have\", \"you'll\": \"you will\", \n       \"you'll've\": \"you will have\", \"you're\": \"you are\", \"you've\": \"you have\" } \n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"59f301722c8826b0cf305ca4f75af4dfce6b8759"},"cell_type":"code","source":"import re, string\nfrom nltk.tokenize import TweetTokenizer    \nfrom nltk.tokenize import word_tokenize\ntokenizer=TweetTokenizer()\ndef clean_text(text):    \n    #fixing apostrope\n    text = text.replace(\"’\", \"'\")\n    #to lower\n    text = text.lower()\n    #remove \\n\n    text = re.sub(\"\\\\n\",\"\",text)\n\n    # remove leaky elements like ip,user\n    text = re.sub(\"\\d{1,3}\\.\\d{1,3}\\.\\d{1,3}\\.\\d{1,3}\",\"\",text)\n    \n    \n    #Split the sentences into words\n    words = tokenizer.tokenize(text)\n    # (')aphostophe  replacement (ie)   you're --> you are  \n    # ( basic dictionary lookup : master dictionary present in a hidden block of code)\n    words = [fill[word] if word in fill else word for word in words]\n    #words = [lem.lemmatize(word, \"v\") for word in words]\n    #words = [i for i in text.split() if i not in eng_stopwords]\n    text = \" \".join(words)\n    return text","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"4547fb78abccca7d99fd0371297360f04948ff62"},"cell_type":"code","source":"# clean the comment_text in train_df\ncleaned_train_comment = []\nfor i in range(0,len(train)):\n    cleaned_comment = clean_text(train['question_text'][i])\n    cleaned_train_comment.append(cleaned_comment)\ntrain['question_text'] = pd.Series(cleaned_train_comment).astype(str)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"3e150184342bc4b0c87cefe15c894dec275dbc64"},"cell_type":"code","source":"# clean the comment_text in train_df\ncleaned_test_comment = []\nfor i in range(0,len(test)):\n    cleaned_comment = clean_text(test['question_text'][i])\n    cleaned_test_comment.append(cleaned_comment)\ntest['question_text'] = pd.Series(cleaned_test_comment).astype(str)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"4658bf95f6e8e97afdee4f68081d888245b6f086"},"cell_type":"code","source":"X_train = train[\"question_text\"].fillna(\"fillna\").values\ny_train = train[\"target\"].values\nX_test = test[\"question_text\"].fillna(\"fillna\").values","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"8792532a288b6387f6f8c250047b4bc387eb7994"},"cell_type":"code","source":"## some config values \nembed_size = 300 # how big is each word vector\nmax_features = 95000 # how many unique words to use (i.e num rows in embedding vector)\nmaxlen = 70 # max number of words in a question to use","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"914bf39e057e2a1edb5d602c0da4d45514ccfaff"},"cell_type":"code","source":"tok = text.Tokenizer(num_words=max_features)\ntok.fit_on_texts(list(X_train) + list(X_test))\nX_train = tok.texts_to_sequences(X_train)\nX_test = tok.texts_to_sequences(X_test)\nx_train = sequence.pad_sequences(X_train, maxlen=maxlen)\nx_test = sequence.pad_sequences(X_test, maxlen=maxlen)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"377d19bc55b33ef4794b74e27d0c1c2769aa1d11"},"cell_type":"code","source":"EMBEDDING_FILE = '../input/embeddings/glove.840B.300d/glove.840B.300d.txt'\ndef get_coefs(word,*arr): return word, np.asarray(arr, dtype='float32')\nembeddings_index = dict(get_coefs(*o.split(\" \")) for o in open(EMBEDDING_FILE))\n\nall_embs = np.stack(embeddings_index.values())\nemb_mean,emb_std = all_embs.mean(), all_embs.std()\nembed_size = all_embs.shape[1]\n\nword_index = tok.word_index\nprint('Found %s unique tokens.' % len(word_index))\nnb_words = min(max_features, len(word_index))\nembedding_matrix = np.random.normal(emb_mean, emb_std, (nb_words, embed_size))\nfor word, i in word_index.items():\n    if i >= max_features: continue\n    embedding_vector = embeddings_index.get(word)\n    if embedding_vector is not None: embedding_matrix[i] = embedding_vector\nprint(\"Embedding matrix Shape : \",embedding_matrix.shape)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e15062a8234c1dba7c62fd1d116f0fb2f91d2a45"},"cell_type":"code","source":"X_tra, X_val, y_tra, y_val = train_test_split(x_train, y_train, train_size=0.95,random_state=123)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"fd58e0ca1ec21c9dacfe4138dc304b4f0979de00"},"cell_type":"code","source":"# https://www.kaggle.com/yekenot/2dcnn-textclassifier\nfrom keras.layers import Input, Embedding, Dense, Conv2D, MaxPool2D\nfrom keras.layers import Reshape, Flatten, Concatenate, Dropout, SpatialDropout1D\nfilter_sizes = [1,2,3,5]\nnum_filters = 36\n\ninp = Input(shape=(maxlen,))\nx = Embedding(max_features, embed_size, weights=[embedding_matrix])(inp)\nx = Reshape((maxlen, embed_size, 1))(x)\n\nmaxpool_pool = []\nfor i in range(len(filter_sizes)):\n    conv = Conv2D(num_filters, kernel_size=(filter_sizes[i], embed_size),\n                                 kernel_initializer='he_normal', activation='elu')(x)\n    maxpool_pool.append(MaxPool2D(pool_size=(maxlen - filter_sizes[i] + 1, 1))(conv))\n\nz = Concatenate(axis=1)(maxpool_pool)   \nz = Flatten()(z)\nz = Dropout(0.1)(z)\n\noutp = Dense(1, activation=\"sigmoid\")(z)\n\nmodel = Model(inputs=inp, outputs=outp)\n# compile the model\n#Adam_opt = Adam(lr=0.0001, beta_1=0.9, beta_2=0.999, epsilon=1e-08, decay=0.0)\n#Adadelta_opt = Adadelta(lr=1.0, rho=0.95, epsilon=None, decay=0.0)\nmodel.compile(loss='binary_crossentropy',optimizer=Adam(lr=0.0001),metrics=['accuracy'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"694c0bd88669ab3b78c87e79c8cf35849b585efe"},"cell_type":"code","source":"model.summary()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"37c503aa559956ee7c4e4ea2be4b7ceb0363b136"},"cell_type":"code","source":"\"\"\"\nfrom keras.utils import plot_model\nplot_model(model, to_file='model.png')\n\nfrom IPython.display import SVG\nfrom keras.utils.vis_utils import model_to_dot\n\nSVG(model_to_dot(model).create(prog='dot', format='svg'))\n\"\"\"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"1d611ca9c7d6077a58e674362b31c30936dcdaa7"},"cell_type":"code","source":"\"\"\" \nclass RocAucEvaluation(Callback):\n    def __init__(self, validation_data=(), interval=1):\n        super(Callback, self).__init__()\n\n        self.interval = interval\n        self.X_val, self.y_val = validation_data\n\n    def on_epoch_end(self, epoch, logs={}):\n        if epoch % self.interval == 0:\n            y_pred = self.model.predict(self.X_val, verbose=0)\n            score = roc_auc_score(self.y_val, y_pred)\n            print(\"\\n ROC-AUC - epoch: {:d} - score: {:.6f}\".format(epoch+1, score))\n            \"\"\"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"52fc6a2b7652572863ebf1fe2eb28767188d4b45"},"cell_type":"code","source":"\"\"\" \nfile_path = \"best_model.hdf5\"\ncheck_point = ModelCheckpoint(file_path, monitor = \"val_loss\", verbose = 1,save_best_only = True, mode = \"min\")\nra_val = RocAucEvaluation(validation_data=(X_val, y_val), interval = 1)\nearly_stop = EarlyStopping(monitor = \"val_loss\", mode = \"min\", patience = 5)\n\"\"\"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"5df32da377285da9afd952fa26ee3d54da78ab62"},"cell_type":"code","source":"history = model.fit(X_tra, y_tra, batch_size = 512, epochs = 2, validation_data = (X_val, y_val), \n                    verbose = 1)#callbacks = [ra_val, check_point, early_stop])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"9c22d728eb70c68db5bd6a75716577e84d53a112"},"cell_type":"code","source":"pred_cnn_val_y = model.predict([X_val], batch_size=1024, verbose=1)\nthresholds = []\nfor thresh in np.arange(0.1, 0.501, 0.01):\n    thresh = np.round(thresh, 2)\n    res = metrics.f1_score(y_val, (pred_cnn_val_y > thresh).astype(int))\n    thresholds.append([thresh, res])\n    print(\"F1 score at threshold {0} is {1}\".format(thresh, res))\n    \nthresholds.sort(key=lambda x: x[1], reverse=True)\nbest_thresh = thresholds[0][0]\nprint(\"Best threshold: \", best_thresh)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"dfcc43bc7a6c77fbb8eab4ffead87da083eef60a"},"cell_type":"code","source":"pred_cnn_test_y = model.predict([x_test], batch_size=1024, verbose=1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"022f4091a1ce6c48db8b18e938307641a3bc1f93"},"cell_type":"code","source":"del word_index, embeddings_index, all_embs, embedding_matrix, model, inp, x\nimport gc; gc.collect()\ntime.sleep(10)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"2089278491329acb2685a0969c9a9dcd6e330d94"},"cell_type":"code","source":"class Attention(Layer):\n    def __init__(self, step_dim,\n                 W_regularizer=None, b_regularizer=None,\n                 W_constraint=None, b_constraint=None,\n                 bias=True, **kwargs):\n        self.supports_masking = True\n        self.init = initializers.get('glorot_uniform')\n\n        self.W_regularizer = regularizers.get(W_regularizer)\n        self.b_regularizer = regularizers.get(b_regularizer)\n\n        self.W_constraint = constraints.get(W_constraint)\n        self.b_constraint = constraints.get(b_constraint)\n\n        self.bias = bias\n        self.step_dim = step_dim\n        self.features_dim = 0\n        super(Attention, self).__init__(**kwargs)\n\n    def build(self, input_shape):\n        assert len(input_shape) == 3\n\n        self.W = self.add_weight((input_shape[-1],),\n                                 initializer=self.init,\n                                 name='{}_W'.format(self.name),\n                                 regularizer=self.W_regularizer,\n                                 constraint=self.W_constraint)\n        self.features_dim = input_shape[-1]\n\n        if self.bias:\n            self.b = self.add_weight((input_shape[1],),\n                                     initializer='zero',\n                                     name='{}_b'.format(self.name),\n                                     regularizer=self.b_regularizer,\n                                     constraint=self.b_constraint)\n        else:\n            self.b = None\n\n        self.built = True\n\n    def compute_mask(self, input, input_mask=None):\n        return None\n\n    def call(self, x, mask=None):\n        features_dim = self.features_dim\n        step_dim = self.step_dim\n\n        eij = K.reshape(K.dot(K.reshape(x, (-1, features_dim)),\n                        K.reshape(self.W, (features_dim, 1))), (-1, step_dim))\n\n        if self.bias:\n            eij += self.b\n\n        eij = K.tanh(eij)\n\n        a = K.exp(eij)\n\n        if mask is not None:\n            a *= K.cast(mask, K.floatx())\n\n        a /= K.cast(K.sum(a, axis=1, keepdims=True) + K.epsilon(), K.floatx())\n\n        a = K.expand_dims(a)\n        weighted_input = x * a\n        return K.sum(weighted_input, axis=1)\n\n    def compute_output_shape(self, input_shape):\n        return input_shape[0],  self.features_dim\n\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"9576025fa7aaea24d121162934a9e386bf9b58ec"},"cell_type":"code","source":"EMBEDDING_FILE = '../input/embeddings/glove.840B.300d/glove.840B.300d.txt'\ndef get_coefs(word,*arr): return word, np.asarray(arr, dtype='float32')\nembeddings_index = dict(get_coefs(*o.split(\" \")) for o in open(EMBEDDING_FILE))\n\nall_embs = np.stack(embeddings_index.values())\nemb_mean,emb_std = all_embs.mean(), all_embs.std()\nembed_size = all_embs.shape[1]\n\nword_index = tok.word_index\nnb_words = min(max_features, len(word_index))\nembedding_matrix = np.random.normal(emb_mean, emb_std, (nb_words, embed_size))\nfor word, i in word_index.items():\n    if i >= max_features: continue\n    embedding_vector = embeddings_index.get(word)\n    if embedding_vector is not None: embedding_matrix[i] = embedding_vector\n        \n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c311c8bcecc531d07454eb0f99c90c79e83a7962"},"cell_type":"code","source":"inp =Input(shape=(maxlen, ))\nx = Embedding(max_features, embed_size, weights=[embedding_matrix], trainable=False)(inp)\nx = Bidirectional(CuDNNGRU(128, kernel_initializer=glorot_normal(seed=12300),return_sequences=True))(x)\nx = Bidirectional(CuDNNGRU(64, kernel_initializer=glorot_normal(seed=12300),return_sequences=True))(x)\nx = Conv1D(128, kernel_size = 3, activation='relu', padding = \"valid\")(x)\nx = Conv1D(128, kernel_size = 3, activation='relu', padding = \"valid\")(x)\nx = Attention(66)(x)\nx = Dense(256, activation='relu')(x)\nx = Dense(256, activation='relu')(x)\nx = Dense(1, activation=\"sigmoid\")(x)\n# compile the model\nmodel = Model(inputs=inp, outputs=x)\n#Adam_opt = Adam(lr=0.0001, beta_1=0.9, beta_2=0.999, epsilon=1e-08, decay=0.0)\n#Adadelta_opt = Adadelta(lr=1.0, rho=0.95, epsilon=None, decay=0.0)\nmodel.compile(loss='binary_crossentropy',optimizer=Adam(lr=0.0001),metrics=['accuracy'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ad9fc9fad042c60ee2364fbcfd78206178ba62d5"},"cell_type":"code","source":"\"\"\"\nfile_path1 = \"best_model1.hdf5\"\ncheck_point = ModelCheckpoint(file_path1, monitor = \"val_loss\", verbose = 1,save_best_only = True, mode = \"min\")\nra_val = RocAucEvaluation(validation_data=(X_val, y_val), interval = 1)\nearly_stop = EarlyStopping(monitor = \"val_loss\", mode = \"min\", patience = 5)\n\"\"\"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"54030e44f00238c13ef6fa404b3af4673147e900"},"cell_type":"code","source":"history = model.fit(X_tra, y_tra, batch_size = 1024, epochs = 4, validation_data = (X_val, y_val), \n                    verbose = 1)#callbacks = [ra_val, check_point, early_stop])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"3b85e1640de0cd27362400d0b9e4da2d3cb2e3ee"},"cell_type":"code","source":"pred_glove_val_y = model.predict([X_val], batch_size=1024, verbose=1)\nfor thresh in np.arange(0.1, 0.501, 0.01):\n    thresh = np.round(thresh, 2)\n    print(\"F1 score at threshold {0} is {1}\".format(thresh, metrics.f1_score(y_val, (pred_glove_val_y>thresh).astype(int))))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"cf23ee0b0bd6dd2bce72414da8f2727992beea23"},"cell_type":"code","source":"pred_glove_test_y = model.predict([x_test], batch_size=1024, verbose=1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"26453e69d33dbe3744dd3e4cc3ae003475b83a4b"},"cell_type":"code","source":"pred_val_y = (3 * pred_glove_val_y  + 2 * pred_cnn_val_y) / 5.0\n\nthresholds = []\nfor thresh in np.arange(0.1, 0.501, 0.01):\n    thresh = np.round(thresh, 2)\n    res = metrics.f1_score(y_val, (pred_val_y > thresh).astype(int))\n    thresholds.append([thresh, res])\n    print(\"F1 score at threshold {0} is {1}\".format(thresh, res))\n    \nthresholds.sort(key=lambda x: x[1], reverse=True)\nbest_thresh = thresholds[0][0]\nprint(\"Best threshold: \", best_thresh)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"18531b1f8a3f8d6d50e0f01f655a4fd377b3a7a8"},"cell_type":"code","source":"pred_test_y = (3 * pred_glove_test_y + 2 * pred_cnn_test_y) / 5.0\npred_test_y = (pred_test_y > best_thresh).astype(int)\nsubmission['prediction'] = pred_test_y\nsubmission.to_csv('submission.csv', index=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"0213840456bf371f2ed64a3aa2fbc4935364515f"},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"077b7c73bd2e29a45d391a801441038ec29072e6"},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}