{"cells":[{"metadata":{"_uuid":"e68c11b4266ce1625d84205028d53fb1b1e6ec62"},"cell_type":"markdown","source":"'mydrop0.2-maxf120-cleannum' の 一番いいアンサンブル で、 my_lstmのsnapshot ensemble\ndropemb0.1のモデルをそれぞれ入れ替える\nmy_lstm_attenを'facal0.1-drop0.15-maxf120-cleannum'に変える\noriginal-cnn1d-glove' を 'facal0.1-drop0.2-maxf120-cleannum'に変える\nconcatで'my_lstm_atten-glove'モデルを変える\nmy-lstmは 最後の2epochをensemble"},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nimport os\nprint(os.listdir(\"../input\"))\n\n# Any results you write to the current directory are saved as output.","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e31d6e126881ee56a1de3efe02fcf309e900ef00"},"cell_type":"code","source":"## some config values \nembed_size = 300 # how big is each word vector\nmax_features = 120000 # how many unique words to use (i.e num rows in embedding vector)\nmaxlen = 70 # max number of words in a question to use","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"522d9790478f62193ea5c315372a2ab9cbe9b27f"},"cell_type":"markdown","source":"**Load packages and data**"},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"import os\nimport re\nimport time\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nfrom tqdm import tqdm\nimport math\nfrom sklearn.model_selection import train_test_split\nfrom sklearn import metrics\n\nfrom keras.preprocessing.text import Tokenizer\nfrom keras.preprocessing.sequence import pad_sequences\nfrom keras.layers import Dense, Input, CuDNNLSTM, Embedding, Dropout, Activation, CuDNNGRU, Conv1D\nfrom keras.layers import Bidirectional, GlobalMaxPool1D, GlobalMaxPooling1D, GlobalAveragePooling1D\nfrom keras.layers import Input, Embedding, Dense, Conv2D, MaxPool2D, concatenate, Conv1D, MaxPool1D\nfrom keras.layers import Reshape, Flatten, Concatenate, Dropout, SpatialDropout1D\nfrom keras.layers import Lambda, Dot\nfrom keras.layers import MaxPool1D, AveragePooling1D\nfrom keras.optimizers import Adam\nfrom keras.models import Model\nfrom keras import backend as K\nfrom keras.engine.topology import Layer\nfrom keras import initializers, regularizers, constraints, optimizers, layers","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"de288c14a79f3f8f416ebf291c81d14aada4da69"},"cell_type":"code","source":"assert len(K.tensorflow_backend._get_available_gpus()) > 0\nK.tensorflow_backend._get_available_gpus()\n\nos.environ['PYTHONHASHSEED'] = '0'\nfrom numpy.random import seed\nseed(1)\nfrom tensorflow import set_random_seed\nset_random_seed(2)\nimport random as rn\nrn.seed(7)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e18ed4339c33036cc39ded79c58901da9fbe0aeb"},"cell_type":"code","source":"\npuncts = [',', '.', '\"', ':', ')', '(', '-', '!', '?', '|', ';', \"'\", '$', '&', '/', '[', ']', '>', '%', '=', '#', '*', '+', '\\\\', '•',  '~', '@', '£', \n '·', '_', '{', '}', '©', '^', '®', '`',  '<', '→', '°', '€', '™', '›',  '♥', '←', '×', '§', '″', '′', 'Â', '█', '½', 'à', '…', \n '“', '★', '”', '–', '●', 'â', '►', '−', '¢', '²', '¬', '░', '¶', '↑', '±', '¿', '▾', '═', '¦', '║', '―', '¥', '▓', '—', '‹', '─', \n '▒', '：', '¼', '⊕', '▼', '▪', '†', '■', '’', '▀', '¨', '▄', '♫', '☆', 'é', '¯', '♦', '¤', '▲', 'è', '¸', '¾', 'Ã', '⋅', '‘', '∞', \n '∙', '）', '↓', '、', '│', '（', '»', '，', '♪', '╩', '╚', '³', '・', '╦', '╣', '╔', '╗', '▬', '❤', 'ï', 'Ø', '¹', '≤', '‡', '√', ]\ndef clean_text(x):\n    x = str(x)\n    for punct in puncts:\n        if punct in x:\n            x = x.replace(punct, f' {punct} ')\n    return x\n\ndef split_text(x):\n    x = wordninja.split(x)\n    return '-'.join(x)\n\ndef clean_numbers(x):\n    x = re.sub('[0-9]{5,}', '#####', x)\n    x = re.sub('[0-9]{4}', '####', x)\n    x = re.sub('[0-9]{3}', '###', x)\n    x = re.sub('[0-9]{2}', '##', x)\n    return x","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"5cdc95950037613c690c49b27930ae0f59eb23c3"},"cell_type":"code","source":"def load_and_prec():\n    train_df = pd.read_csv(\"../input/train.csv\")\n    test_df = pd.read_csv(\"../input/test.csv\")\n    \n    train_df[\"question_text\"] = train_df[\"question_text\"].str.lower()\n    test_df[\"question_text\"] = test_df[\"question_text\"].str.lower()\n    \n    train_df[\"question_text\"] = train_df[\"question_text\"].apply(lambda x: clean_text(x))\n    test_df[\"question_text\"] = test_df[\"question_text\"].apply(lambda x: clean_text(x))\n    \n    train_df[\"question_text\"] = train_df[\"question_text\"].apply(lambda x: clean_numbers(x))\n    test_df[\"question_text\"] = test_df[\"question_text\"].apply(lambda x: clean_numbers(x))\n    \n    print(\"Train shape : \",train_df.shape)\n    print(\"Test shape : \",test_df.shape)\n\n    ## fill up the missing values\n    train_X = train_df[\"question_text\"].fillna(\"_##_\").values\n    test_X = test_df[\"question_text\"].fillna(\"_##_\").values\n\n    ## Tokenize the sentences\n    tokenizer = Tokenizer(num_words=max_features)\n    tokenizer.fit_on_texts(list(train_X))\n    train_X = tokenizer.texts_to_sequences(train_X)\n    test_X = tokenizer.texts_to_sequences(test_X)\n\n    ## Pad the sentences \n    train_X = pad_sequences(train_X, maxlen=maxlen)\n    test_X = pad_sequences(test_X, maxlen=maxlen)\n\n    ## Get the target values\n    train_y = train_df['target'].values\n    \n    #shuffling the data\n    np.random.seed(2018)\n    trn_idx = np.random.permutation(len(train_X))\n\n    train_X = train_X[trn_idx]\n    train_y = train_y[trn_idx]\n    \n    return train_X, test_X, train_y, tokenizer.word_index, test_df['qid']","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"dba1893c267a1e7536bbf720636647d85c7e349c"},"cell_type":"markdown","source":"**Load embeddings**"},{"metadata":{"trusted":true,"_uuid":"a662716cc5fbbcc0c84019a87c52332ed8912e8d"},"cell_type":"code","source":"def load_glove(word_index):\n    EMBEDDING_FILE = '../input/embeddings/glove.840B.300d/glove.840B.300d.txt'\n    def get_coefs(word,*arr): return word, np.asarray(arr, dtype='float32')\n    embeddings_index = dict(get_coefs(*o.split(\" \")) for o in open(EMBEDDING_FILE))\n\n    all_embs = np.stack(embeddings_index.values())\n    emb_mean,emb_std = -0.005838499,0.48782197\n    embed_size = all_embs.shape[1]\n\n    # word_index = tokenizer.word_index\n    nb_words = min(max_features, len(word_index))\n    embedding_matrix = np.random.normal(emb_mean, emb_std, (nb_words, embed_size))\n    for word, i in word_index.items():\n        if i >= max_features: continue\n        embedding_vector = embeddings_index.get(word)\n        if embedding_vector is not None: embedding_matrix[i] = embedding_vector\n            \n    return embedding_matrix \n    \ndef load_fasttext(word_index):    \n    EMBEDDING_FILE = '../input/embeddings/wiki-news-300d-1M/wiki-news-300d-1M.vec'\n    def get_coefs(word,*arr): return word, np.asarray(arr, dtype='float32')\n    embeddings_index = dict(get_coefs(*o.split(\" \")) for o in open(EMBEDDING_FILE) if len(o)>100)\n\n    all_embs = np.stack(embeddings_index.values())\n    emb_mean,emb_std = all_embs.mean(), all_embs.std()\n    embed_size = all_embs.shape[1]\n\n    # word_index = tokenizer.word_index\n    nb_words = min(max_features, len(word_index))\n    embedding_matrix = np.random.normal(emb_mean, emb_std, (nb_words, embed_size))\n    for word, i in word_index.items():\n        if i >= max_features: continue\n        embedding_vector = embeddings_index.get(word)\n        if embedding_vector is not None: embedding_matrix[i] = embedding_vector\n\n    return embedding_matrix\n\ndef load_para(word_index):\n    EMBEDDING_FILE = '../input/embeddings/paragram_300_sl999/paragram_300_sl999.txt'\n    def get_coefs(word,*arr): return word, np.asarray(arr, dtype='float32')\n    embeddings_index = dict(get_coefs(*o.split(\" \")) for o in open(EMBEDDING_FILE, encoding=\"utf8\", errors='ignore') if len(o)>100)\n\n    all_embs = np.stack(embeddings_index.values())\n    emb_mean,emb_std = -0.0053247833,0.49346462\n    embed_size = all_embs.shape[1]\n    print(emb_mean,emb_std,\"para\")\n\n    # word_index = tokenizer.word_index\n    nb_words = min(max_features, len(word_index))\n    embedding_matrix = np.random.normal(emb_mean, emb_std, (nb_words, embed_size))\n    for word, i in word_index.items():\n        if i >= max_features: continue\n        embedding_vector = embeddings_index.get(word)\n        if embedding_vector is not None: embedding_matrix[i] = embedding_vector\n    \n    return embedding_matrix","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"2da05ee1bb9912932b695fc622b6bcc7774bde70"},"cell_type":"code","source":"def focal_loss(y_true, y_pred):\n    gamma = 0.1\n    true_loss = K.mean(y_true * ((1-y_pred)**gamma * K.log(y_pred + K.epsilon())), axis=0)\n    false_loss = K.mean((1-y_true) * (y_pred**gamma * K.log(1-y_pred + K.epsilon())), axis=0)\n    return - (true_loss + false_loss)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"73d68544af4c48bf9ee37492ecd05feb0b494351"},"cell_type":"markdown","source":"**CNN Model**"},{"metadata":{"trusted":true,"_uuid":"37d26d217c886d24ecb7b187422ba5804684c5a6"},"cell_type":"code","source":"# https://www.kaggle.com/yekenot/2dcnn-textclassifier\ndef model_cnn_1d(embedding_matrix):\n    filter_sizes = [1,2,3,5]\n    num_filters = 36\n\n    inp = Input(shape=(maxlen,)) # batch sizeを含まない\n    # maxlen: 70\n    # max_features: indexの数\n    # embed_size: embedの次元 300 先頭できまってる\n    x = Embedding(max_features, embed_size, weights=[embedding_matrix])(inp) # (batch, embeddim, input_length)\n    x = Reshape((maxlen, embed_size))(x) # batch, input_length, embed_size\n\n    maxpool_pool = []\n    for i in range(len(filter_sizes)):\n        conv = Conv1D(num_filters, kernel_size=filter_sizes[i],\n                                     kernel_initializer='he_normal', activation='elu')(x)\n        maxpool_pool.append(MaxPool1D(pool_size=maxlen - filter_sizes[i] + 1)(conv))\n\n    z = Concatenate(axis=1)(maxpool_pool)   \n    z = Flatten()(z)\n    z = Dropout(0.1)(z)\n\n    outp = Dense(1, activation=\"sigmoid\")(z)\n\n    model = Model(inputs=inp, outputs=outp)\n    model.compile(loss=focal_loss, optimizer='adam', metrics=['accuracy'])\n    \n    return model","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"5a676c3a275514a3351edf306e02d832a5f39317"},"cell_type":"markdown","source":"**Attention layer**"},{"metadata":{"trusted":true,"_uuid":"84e00df2c7b94205f5588af503f62412c48f46f3"},"cell_type":"code","source":"# https://www.kaggle.com/suicaokhoailang/lstm-attention-baseline-0-652-lb\n\nclass Attention(Layer):\n    def __init__(self, step_dim,\n                 W_regularizer=None, b_regularizer=None,\n                 W_constraint=None, b_constraint=None,\n                 bias=True, **kwargs):\n        self.supports_masking = True\n        self.init = initializers.get('glorot_uniform')\n\n        self.W_regularizer = regularizers.get(W_regularizer)\n        self.b_regularizer = regularizers.get(b_regularizer)\n\n        self.W_constraint = constraints.get(W_constraint)\n        self.b_constraint = constraints.get(b_constraint)\n\n        self.bias = bias\n        self.step_dim = step_dim\n        self.features_dim = 0\n        super(Attention, self).__init__(**kwargs)\n\n    def build(self, input_shape):\n        assert len(input_shape) == 3\n\n        self.W = self.add_weight((input_shape[-1],),\n                                 initializer=self.init,\n                                 name='{}_W'.format(self.name),\n                                 regularizer=self.W_regularizer,\n                                 constraint=self.W_constraint)\n        self.features_dim = input_shape[-1]\n\n        if self.bias:\n            self.b = self.add_weight((input_shape[1],),\n                                     initializer='zero',\n                                     name='{}_b'.format(self.name),\n                                     regularizer=self.b_regularizer,\n                                     constraint=self.b_constraint)\n        else:\n            self.b = None\n\n        self.built = True\n\n    def compute_mask(self, input, input_mask=None):\n        return None\n\n    def call(self, x, mask=None):\n        features_dim = self.features_dim\n        step_dim = self.step_dim\n\n        eij = K.reshape(K.dot(K.reshape(x, (-1, features_dim)),\n                        K.reshape(self.W, (features_dim, 1))), (-1, step_dim))\n\n        if self.bias:\n            eij += self.b\n\n        a = K.exp(eij)\n\n        if mask is not None:\n            a *= K.cast(mask, K.floatx())\n\n        a /= K.cast(K.sum(a, axis=1, keepdims=True) + K.epsilon(), K.floatx())\n\n        a = K.expand_dims(a)\n        weighted_input = x * a\n        return K.sum(weighted_input, axis=1)\n\n    def compute_output_shape(self, input_shape):\n        return input_shape[0],  self.features_dim","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d96793d88c22274d985436e192f62970c227c324"},"cell_type":"markdown","source":"**LSTM models**"},{"metadata":{"trusted":true,"_uuid":"05164d541a0c35cae727d0338548d156efe21427"},"cell_type":"code","source":"def model_lstm_atten_mix(embedding_matrix):\n    inp = Input(shape=(maxlen,))\n    x = Embedding(max_features, embed_size, weights=[embedding_matrix], trainable=False)(inp)\n    x = Bidirectional(CuDNNLSTM(128, return_sequences=True))(x)\n    x = Bidirectional(CuDNNLSTM(64, return_sequences=True))(x)\n    x = Attention(maxlen)(x)\n    x = Dense(64, activation=\"elu\")(x)\n    x = Dense(1, activation=\"sigmoid\")(x)\n    model = Model(inputs=inp, outputs=x)\n    model.compile(loss='binary_crossentropy', optimizer='adam', metrics=['accuracy'])\n    \n    return model\n\ndef model_lstm_atten_glove(embedding_matrix):\n    inp = Input(shape=(maxlen,))\n    x = Embedding(max_features, embed_size, weights=[embedding_matrix], trainable=False)(inp)\n    x = Dropout(0.1)(x)\n    x = Bidirectional(CuDNNLSTM(128, return_sequences=True))(x)\n    x = Bidirectional(CuDNNLSTM(64, return_sequences=True))(x)\n    x = Attention(maxlen)(x)\n    x = Dense(64, activation=\"relu\")(x)\n    x = Dense(1, activation=\"sigmoid\")(x)\n    model = Model(inputs=inp, outputs=x)\n    model.compile(loss='binary_crossentropy', optimizer='adam', metrics=['accuracy'])\n    \n    return model","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"17b13ef39fbbf1919307c23efd516eddc2135023"},"cell_type":"code","source":"def model_lstm_du(embedding_matrix):\n    inp = Input(shape=(maxlen,))\n    x = Embedding(max_features, embed_size, weights=[embedding_matrix])(inp)\n    x = Bidirectional(CuDNNGRU(64, return_sequences=True))(x)\n    avg_pool = GlobalAveragePooling1D()(x)\n    max_pool = GlobalMaxPooling1D()(x)\n    conc = concatenate([avg_pool, max_pool])\n    conc = Dense(64, activation=\"relu\")(conc)\n    conc = Dropout(0.1)(conc)\n    outp = Dense(1, activation=\"sigmoid\")(conc)\n    \n    model = Model(inputs=inp, outputs=outp)\n    model.compile(loss='binary_crossentropy', optimizer='adam', metrics=['accuracy'])\n    return model","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"1b9daf47855461cec25f3bf4f51ec8a8bd990f32"},"cell_type":"code","source":"def model_my_lstm(embedding_matrix):\n    p_drop = 0.1\n\n    inp = Input(shape=(maxlen,))\n    x = Embedding(max_features, embed_size*2, weights=[embedding_matrix], trainable=False)(inp)\n    x = Dropout(p_drop)(x)\n\n    x = Bidirectional(CuDNNLSTM(128, return_sequences=True))(x)\n    x = Dropout(p_drop)(x)\n\n    x = Bidirectional(CuDNNLSTM(64, return_sequences=True))(x)\n    \n    atten_x = Attention(maxlen)(x)\n        \n    max_x = MaxPool1D(maxlen)(x) # 1, units\n    max_x = Flatten()(max_x) # units\n    \n    ave_x = AveragePooling1D(maxlen)(x) # 1, units\n    ave_x = Flatten()(ave_x) # units\n    \n    x = Concatenate()([atten_x, max_x, ave_x]) # units * 6\n    x = Dense(1, activation=\"sigmoid\")(x)\n    model = Model(inputs=inp, outputs=x)\n    model.compile(loss=focal_loss, optimizer='adam', metrics=['accuracy'])\n    \n    return model","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"a8c857424e9c9f1703a71c1c0ade28713314dd29"},"cell_type":"markdown","source":"**Train and predict**"},{"metadata":{"trusted":true,"_uuid":"e8523d876b6eae762e673b777cc7af4d7f085792"},"cell_type":"code","source":"# https://www.kaggle.com/strideradu/word2vec-and-gensim-go-go-go\ndef train_pred(model, epochs):\n    for e in range(epochs):\n        model.fit(train_X, train_y, batch_size=512, epochs=1, verbose=2)\n    pred_test_y = model.predict([test_X], batch_size=1024, verbose=0)\n    return pred_test_y\n\ndef train_pred_snapshot(model, epochs, start_epoch):\n    pred_list = []\n    for e in range(epochs):\n        model.fit(train_X, train_y, batch_size=512, epochs=1, verbose=2)\n        if e >= start_epoch:\n            pred_test_y = model.predict([test_X], batch_size=1024, verbose=0)\n            pred_list.append(pred_test_y)\n    \n    return pred_list","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"f79081928ca032fbfe3b90c6d3ce91cf57d443d8"},"cell_type":"markdown","source":"**Main part: load, train, pred and blend**"},{"metadata":{"trusted":true,"_uuid":"99d03d2eb63600f1b222522616eab3fa35819f37"},"cell_type":"code","source":"%%time\ntrain_X, test_X, train_y, word_index, test_df = load_and_prec()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"7d4ee0793f5a92bd4e845ff1b765965b2b7171e2"},"cell_type":"code","source":"%%time\nembedding_matrix_glove = load_glove(word_index)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d068710ef4dbc38599832b19b0768b7cc6cb5d31"},"cell_type":"code","source":"%%time\nembedding_matrix_para = load_para(word_index)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f0ea9b1468bd7cd3ceead2593c641900dc3a2a77"},"cell_type":"code","source":"embedding_matrix_mix = np.mean([embedding_matrix_glove, embedding_matrix_para], axis = 0)\nembedding_matrix_concat = np.concatenate([embedding_matrix_glove, embedding_matrix_para], axis = 1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"8b0ce7a68d32a893affb0f3e57e0d1517776e1b1"},"cell_type":"code","source":"%%time\noutputs = []\n\npred_test_y = train_pred(model_cnn_1d(embedding_matrix_glove), epochs = 2) # GloVe only\noutputs.append([pred_test_y, '1d-CNN'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"2d5802cdc286219cbddd4caa9851d08cdf39ebf6"},"cell_type":"code","source":"%%time\npred_test_y = train_pred(model_lstm_du(embedding_matrix_mix), epochs = 2)\noutputs.append([pred_test_y, 'LSTM-DU'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"cbf875ed7432fb18fc76bad6111e4a73ff40aeab"},"cell_type":"code","source":"%%time\npred_test_y = train_pred(model_lstm_atten_mix(embedding_matrix_mix), epochs = 3)\noutputs.append([pred_test_y, '2-LSTM-attention-mix'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"151316f426f9dc99857795970874d2c1aef044bb"},"cell_type":"code","source":"%%time\npred_test_y = train_pred(model_lstm_atten_glove(embedding_matrix_glove), epochs = 3)\noutputs.append([pred_test_y, '2-LSTM-attention-glov'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"af0a04afe94d815a02d7c1817e4e649a26e55990"},"cell_type":"code","source":"%%time\npred_test_y_list = train_pred_snapshot(model_my_lstm(embedding_matrix_concat), epochs=5, start_epoch=3)\noutputs.extend([[pred, 'my_lstm'] for pred in pred_test_y_list])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"af06a54cf873dd2eacd7953894e4442fddf25257"},"cell_type":"code","source":"%%time\nw_list = [0.15573309, 0.15155236, 0.20362984, 0.1206986,  0.14503201, 0.2233541]\nw_normed = np.array(w_list) / np.sum(w_list)\npred_test_y = np.sum([outputs[i][0]*w_normed[i] for i in range(len(w_normed))], axis = 0)\npred_test_y = (pred_test_y > 0.3434).astype(int)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"74a84952d3a6a2b265dcd72a1ba8f41e315df332"},"cell_type":"code","source":"%%time\nout_df = pd.DataFrame({\"qid\":test_df.values})\nout_df['prediction'] = pred_test_y\nout_df.to_csv(\"submission.csv\", index=False)","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}