{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\nfrom multiprocessing import Pool\nimport os\nprint(os.listdir(\"../input\"))\n\n# Any results you write to the current directory are saved as output.\nimport random\nfrom tqdm import tqdm\n\nfrom sklearn.model_selection import train_test_split\nfrom sklearn import metrics\nfrom gensim.models import KeyedVectors\nimport torch\n\nfrom keras.preprocessing.text import Tokenizer\nfrom keras.preprocessing.sequence import pad_sequences\nfrom keras.layers import Dense, Input, LSTM, Embedding, Dropout, Activation, Conv1D, GRU, CuDNNGRU, CuDNNLSTM, BatchNormalization\nfrom keras.layers import Bidirectional, GlobalMaxPool1D, MaxPool1D, Add, Flatten, Layer\nfrom keras.layers import GlobalAveragePooling1D, GlobalMaxPooling1D, concatenate, SpatialDropout1D, add, Reshape\nfrom keras.models import Model, load_model\nfrom keras import initializers, regularizers, constraints, optimizers, layers, callbacks, Sequential\nfrom keras import backend as k\nfrom keras.engine import InputSpec, Layer\nfrom keras.optimizers import Adam\nfrom keras.callbacks import *\n\nimport gensim\nfrom gensim.models import Word2Vec\nfrom textblob import TextBlob\nimport re\n\nfrom nltk.stem import PorterStemmer\nfrom nltk.stem.lancaster import LancasterStemmer\nfrom nltk.stem import SnowballStemmer","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"ps = PorterStemmer()\nls = LancasterStemmer()\nss = SnowballStemmer('english')","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"train = pd.read_csv('../input/jigsaw-unintended-bias-in-toxicity-classification/train.csv')\ntest = pd.read_csv('../input/jigsaw-unintended-bias-in-toxicity-classification/test.csv')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"identity_columns = ['male', 'female', 'homosexual_gay_or_lesbian', 'christian', 'jewish', 'muslim', 'black', 'white',\n                    'psychiatric_or_mental_illness']\nfor col in identity_columns+['target']:\n    train[col] = np.where(train[col]>=0.5, True, False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def seed_everything(seed=1234):\n    random.seed(seed)\n    os.environ['PYTHONHASHSEED'] = str(seed)\n    np.random.seed(seed)\n    torch.manual_seed(seed)\n    torch.cuda.manual_seed(seed)\n    torch.backends.cudnn.deterministic = True\n    from tensorflow import set_random_seed\n    set_random_seed(2)\n\nseed_everything()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"**Processing**"},{"metadata":{"trusted":true},"cell_type":"code","source":"# Define some Global Variables\nmax_features = 150000 # Maximum Number of words we want to include in our dictionary\nmaxlen = 256 # No of words in question we want to create a sequence with\nembed_size = 302# Size of word to vec embedding we are using","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Some preprocesssing that will be common to all the text classification methods you will see. \npuncts = [',', '.', '\"', ':', ')', '(', '-', '!', '?', '|', ';', \"'\", '$', '&', '/', '[', ']', '>', '%', '=', '#', '*', '+', '\\\\', '•',  '~', '@', '£', \n '·', '_', '{', '}', '©', '^', '®', '`',  '<', '→', '°', '€', '™', '›',  '♥', '←', '×', '§', '″', '′', 'Â', '█', '½', 'à', '…', \n '“', '★', '”', '–', '●', 'â', '►', '−', '¢', '²', '¬', '░', '¶', '↑', '±', '¿', '▾', '═', '¦', '║', '―', '¥', '▓', '—', '‹', '─', \n '▒', '：', '¼', '⊕', '▼', '▪', '†', '■', '’', '▀', '¨', '▄', '♫', '☆', 'é', '¯', '♦', '¤', '▲', 'è', '¸', '¾', 'Ã', '⋅', '‘', '∞', \n '∙', '）', '↓', '、', '│', '（', '»', '，', '♪', '╩', '╚', '³', '・', '╦', '╣', '╔', '╗', '▬', '❤', 'ï', 'Ø', '¹', '≤', '‡', '√', ]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def clean_text(x):\n    x = str(x)\n    for punct in puncts:\n        x = x.replace(punct, f' {punct} ')\n    return x","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def clean_numbers(x):\n    x = re.sub('[0-9]{5,}', '#####', x)\n    x = re.sub('[0-9]{4}', '####', x)\n    x = re.sub('[0-9]{3}', '###', x)\n    x = re.sub('[0-9]{2}', '##', x)\n    return x\n    ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"contraction_mapping = {\"ain't\": \"is not\", \"aren't\": \"are not\",\"can't\": \"cannot\", \"'cause\": \"because\", \"could've\": \"could have\", \"couldn't\": \"could not\", \"didn't\": \"did not\",  \"doesn't\": \"does not\", \"don't\": \"do not\", \"hadn't\": \"had not\", \"hasn't\": \"has not\", \"haven't\": \"have not\", \"he'd\": \"he would\",\"he'll\": \"he will\", \"he's\": \"he is\", \"how'd\": \"how did\", \"how'd'y\": \"how do you\", \"how'll\": \"how will\", \"how's\": \"how is\",  \"I'd\": \"I would\", \"I'd've\": \"I would have\", \"I'll\": \"I will\", \"I'll've\": \"I will have\",\"I'm\": \"I am\", \"I've\": \"I have\", \"i'd\": \"i would\", \"i'd've\": \"i would have\", \"i'll\": \"i will\",  \"i'll've\": \"i will have\",\"i'm\": \"i am\", \"i've\": \"i have\", \"isn't\": \"is not\", \"it'd\": \"it would\", \"it'd've\": \"it would have\", \"it'll\": \"it will\", \"it'll've\": \"it will have\",\"it's\": \"it is\", \"let's\": \"let us\", \"ma'am\": \"madam\", \"mayn't\": \"may not\", \"might've\": \"might have\",\"mightn't\": \"might not\",\"mightn't've\": \"might not have\", \"must've\": \"must have\", \"mustn't\": \"must not\", \"mustn't've\": \"must not have\", \"needn't\": \"need not\", \"needn't've\": \"need not have\",\"o'clock\": \"of the clock\", \"oughtn't\": \"ought not\", \"oughtn't've\": \"ought not have\", \"shan't\": \"shall not\", \"sha'n't\": \"shall not\", \"shan't've\": \"shall not have\", \"she'd\": \"she would\", \"she'd've\": \"she would have\", \"she'll\": \"she will\", \"she'll've\": \"she will have\", \"she's\": \"she is\", \"should've\": \"should have\", \"shouldn't\": \"should not\", \"shouldn't've\": \"should not have\", \"so've\": \"so have\",\"so's\": \"so as\", \"this's\": \"this is\",\"that'd\": \"that would\", \"that'd've\": \"that would have\", \"that's\": \"that is\", \"there'd\": \"there would\", \"there'd've\": \"there would have\", \"there's\": \"there is\", \"here's\": \"here is\",\"they'd\": \"they would\", \"they'd've\": \"they would have\", \"they'll\": \"they will\", \"they'll've\": \"they will have\", \"they're\": \"they are\", \"they've\": \"they have\", \"to've\": \"to have\", \"wasn't\": \"was not\", \"we'd\": \"we would\", \"we'd've\": \"we would have\", \"we'll\": \"we will\", \"we'll've\": \"we will have\", \"we're\": \"we are\", \"we've\": \"we have\", \"weren't\": \"were not\", \"what'll\": \"what will\", \"what'll've\": \"what will have\", \"what're\": \"what are\",  \"what's\": \"what is\", \"what've\": \"what have\", \"when's\": \"when is\", \"when've\": \"when have\", \"where'd\": \"where did\", \"where's\": \"where is\", \"where've\": \"where have\", \"who'll\": \"who will\", \"who'll've\": \"who will have\", \"who's\": \"who is\", \"who've\": \"who have\", \"why's\": \"why is\", \"why've\": \"why have\", \"will've\": \"will have\", \"won't\": \"will not\", \"won't've\": \"will not have\", \"would've\": \"would have\", \"wouldn't\": \"would not\", \"wouldn't've\": \"would not have\", \"y'all\": \"you all\", \"y'all'd\": \"you all would\",\"y'all'd've\": \"you all would have\",\"y'all're\": \"you all are\",\"y'all've\": \"you all have\",\"you'd\": \"you would\", \"you'd've\": \"you would have\", \"you'll\": \"you will\", \"you'll've\": \"you will have\", \"you're\": \"you are\", \"you've\": \"you have\" }","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def clean_contractions(text, mapping):\n    specials = [\"’\", \"‘\", \"´\", \"`\"]\n    for s in specials:\n        text = text.replace(s, \"'\")\n    text = ' '.join([mapping[t] if t in mapping else t for t in text.split(\" \")])\n    return text","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def load_and_proc():\n    # split into train and validation\n    train['comment_text'] = train['comment_text'].apply(lambda x : clean_text(x))\n    test['comment_text'] = test['comment_text'].apply(lambda x : clean_text(x))\n    train['comment_text'] = train['comment_text'].apply(lambda x : clean_numbers(x))\n    test['comment_text'] = test['comment_text'].apply(lambda x : clean_numbers(x))\n    train['comment_text'] = train['comment_text'].apply(lambda x : clean_contractions(x,contraction_mapping))\n    test['comment_text'] = test['comment_text'].apply(lambda x : clean_contractions(x,contraction_mapping))\n    \n    df_train, df_valid = train_test_split(train, test_size=0.33)\n    df_test = test\n    \n    df_train.loc[:,'set_'] = 'train'\n    df_valid.loc[:,'set_'] = 'valid'\n    df_test.loc[:,'set_'] = 'test'\n\n    set_indices = df_train.loc[:,'set_']\n    set_indices.append(df_valid.loc[:,'set_'])\n    set_indices.append(df_test.loc[:,'set_'])\n\n    y_train = np.asarray(df_train['target'])\n    y_valid = np.asarray(df_valid['target'])\n\n    set_indices_label = df_train.loc[:,'set_']\n    set_indices_label = set_indices_label.append(df_valid.loc[:,'set_'])\n    \n    X_train = df_train['comment_text'].fillna('_##_').values\n    X_val = df_valid['comment_text'].fillna('_##_').values\n    X_test = df_test['comment_text'].fillna('_##_').values\n    \n    all_text = list(X_train)\n    all_text.append(list(X_val))\n    all_text.append(list(X_test))\n    \n    tokenizer = Tokenizer(num_words=max_features)\n    tokenizer.fit_on_texts(all_text)\n    \n    X_train = tokenizer.texts_to_sequences(X_train)\n    X_val = tokenizer.texts_to_sequences(X_val)\n    X_test = tokenizer.texts_to_sequences(X_test)\n    \n    X_train = pad_sequences(X_train, maxlen=maxlen)\n    X_val = pad_sequences(X_val, maxlen=maxlen)\n    X_test = pad_sequences(X_test, maxlen=maxlen)\n    \n    #shuffling the data\n    np.random.seed(2019)\n    train_idx = np.random.permutation(len(X_train))\n    valid_idx = np.random.permutation(len(X_val))\n    print(type(X_train))\n    print(type(y_train))\n    X_train = X_train[train_idx]\n    X_val = X_val[valid_idx]\n    Y_train = y_train[train_idx]\n    Y_val = y_valid[valid_idx]\n    \n    return X_train, X_val, X_test, Y_train, Y_val, tokenizer.word_index","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"X_train, X_val,X_test, Y_train, Y_val, word_index = load_and_proc()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#word_index.items()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import gc\ngc.collect()\ndel train\ndel test","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def load_glove():\n    EMBEDDING_FILE = '../input/emb-model/glove.840B.300d.txt'\n    def get_coef(word, *arr): return word, np.asarray(arr, dtype=np.float32)[:300]\n    embeddings_index = dict(get_coef(*o.split(\" \")) for o in open(EMBEDDING_FILE))\n    \n    return embeddings_index","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def load_paragram():\n    EMBEDDING_FILE = '../input/paragram-300-sl999/paragram_300_sl999.txt'\n    def get_coef(word, *arr): return word, np.asarray(arr, dtype=np.float32)[:300]\n    embeddings_index = dict(get_coef(*o.split(\" \")) for o in open(EMBEDDING_FILE, errors='ignore', encoding='utf-8'))\n    \n    return embeddings_index","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def load_fasttext():\n    EMBEDDING_FILE = '../input/fasttext-wiki-news-300d-1m/wiki-news-300d-1M.vec'\n    def get_coef(word, *arr): return word, np.asarray(arr, dtype=np.float32)\n    embeddings_index = dict(get_coef(*o.split(\" \")) for o in open(EMBEDDING_FILE) if len(o) > 100)\n    \n    return embeddings_index","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def create_emb(word_index, embeddings_index):\n    #embedding_index_paragram = load_paragram()\n    #embeddings_index = load_glove()\n    #all_embs_glove = np.stack(embeddings_index.values())\n    #all_embs_paragram = np.stack(embedding_index_paragram.values())\n    emb_mean,emb_std = -0.005838499,0.48782197\n    #final_emb = np.concatenate([all_embs_glove,all_embs_paragram ])\n    #embed_size = all_embs_glove.shape[1]\n    nb_words = min(max_features, len(word_index))\n    #embeddings_index.update(embedding_index_paragram)\n    embedding_matrix = np.random.normal(emb_mean, emb_std, (nb_words, embed_size))\n    count_found = nb_words\n    for word, i in tqdm(word_index.items()):\n        if i >= max_features: continue\n        embedding_vector = embeddings_index.get(word)\n        word_sent = TextBlob(word).sentiment\n        # Extra information we are passing to our embeddings\n        extra_embed = [word_sent.polarity, word_sent.subjectivity]\n        if embedding_vector is not None:\n            embedding_matrix[i] = np.append(embedding_vector, extra_embed)\n            continue\n        key = word.lower()\n        embedding_vector = embeddings_index.get(key)\n        if embedding_vector is not None:\n            embedding_matrix[i] = np.append(embedding_vector, extra_embed)\n            continue\n        key = word.upper()\n        embedding_vector = embeddings_index.get(key)\n        if embedding_vector is not None:\n            embedding_matrix[i] = np.append(embedding_vector, extra_embed)\n            continue\n        key = word.capitalize()\n        embedding_vector = embeddings_index.get(key)\n        if embedding_vector is not None:\n            embedding_matrix[i] = np.append(embedding_vector, extra_embed)\n            continue\n        key = ps.stem(word)\n        embedding_vector = embeddings_index.get(key)\n        if embedding_vector is not None:\n            embedding_matrix[i] = np.append(embedding_vector, extra_embed)\n            continue\n        key = ls.stem(word)\n        embedding_vector = embeddings_index.get(key)\n        if embedding_vector is not None:\n            embedding_matrix[i] = np.append(embedding_vector, extra_embed)\n            continue\n        key = ss.stem(word)\n        embedding_vector = embeddings_index.get(key)\n        if embedding_vector is not None:\n            embedding_matrix[i] = np.append(embedding_vector, extra_embed)\n            continue\n        embedding_matrix[i,300:] = extra_embed\n        count_found -= 1\n    print('Total words ', nb_words)\n    print(\"Got embedding for \",count_found,\" words.\")\n    return embedding_matrix","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#np.stack(embeddings_index.values())\ngc.collect()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#paragram_embedding = create_emb(word_index,load_paragram())\nglove_embedding = create_emb(word_index,load_glove())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#final_matrix = np.concatenate([glove_embedding,paragram_embedding], 1)\nfinal_matrix = glove_embedding","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"gc.collect()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def dot_product(x, kernel):\n    \"\"\"\n    Wrapper for dot product operation, in order to be compatible with both\n    Theano and Tensorflow\n    Args:\n        x (): input\n        kernel (): weights\n    Returns:\n    \"\"\"\n    if K.backend() == 'tensorflow':\n        return K.squeeze(K.dot(x, K.expand_dims(kernel)), axis=-1)\n    else:\n        return K.dot(x, kernel)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"class Attention(Layer):\n    \n    def __init__(self, \n                W_regulariser=None,b_regulariser=None,u_regulariser=None,\n                W_constraint=None,b_constraint=None,u_constraint=None,\n                biase=True, **kwargs):\n        self.support_masking = True\n        self.initializer = initializers.get('glorot_uniform')\n        \n        self.W_regulariser = regularizers.get(W_regulariser)\n        self.b_regulariser = regularizers.get(b_regulariser)\n        self.u_regulariser = regularizers.get(u_regulariser)\n        \n        self.W_constraint = constraints.get(W_constraint)\n        self.b_constraint = constraints.get(b_constraint)\n        self.u_constraint = constraints.get(u_constraint)\n        \n        self.biase = biase\n        super(Attention,self).__init__(**kwargs)\n        \n    def build(self, input_shape):\n        assert len(input_shape) == 3\n        self.W = self.add_weight(shape=(input_shape[-1],input_shape[-1],),\n                                 initializer=self.initializer,\n                                 name='{}_W'.format(self.name),\n                                 regularizer=self.W_regulariser,\n                                 constraint=self.W_constraint)\n        if self.biase:\n            self.b =  self.add_weight(shape=(input_shape[-1],),\n                                      initializer = 'zero',\n                                      name='{}_b'.format(self.name),\n                                      regularizer=self.b_regulariser,\n                                      constraint=self.b_constraint\n                                     )\n        self.u = self.add_weight(shape=(input_shape[-1],),\n                                 initializer=self.initializer,\n                                 name='{}_u'.format(self.name),\n                                 regularizer=self.u_regulariser,\n                                 constraint=self.u_constraint)\n        super(Attention, self).build(input_shape)\n    \n    def compute_mask(self, input, input_mask=None):\n        # do not pass the mask to the next layers\n        return None\n    \n    def call(self, x, mask=None):\n        uit = dot_product(x, self.W)\n\n        if self.biase:\n            uit += self.b\n\n        uit = K.tanh(uit)\n        ait = dot_product(uit, self.u)\n\n        a = K.exp(ait)\n\n        # apply mask after the exp. will be re-normalized next\n        if mask is not None:\n            # Cast the mask to floatX to avoid float64 upcasting in theano\n            a *= K.cast(mask, K.floatx())\n\n        # in some cases especially in the early stages of training the sum may be almost zero\n        # and this results in NaN's. A workaround is to add a very small positive number ε to the sum.\n        # a /= K.cast(K.sum(a, axis=1, keepdims=True), K.floatx())\n        a /= K.cast(K.sum(a, axis=1, keepdims=True) + K.epsilon(), K.floatx())\n\n        a = K.expand_dims(a)\n        weighted_input = x * a\n        return K.sum(weighted_input, axis=1)\n\n    def compute_output_shape(self, input_shape):\n        return input_shape[0], input_shape[-1]\n        ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def model_lstm_du(final_embedding):\n    inp = Input(shape=(maxlen,))\n    x = Embedding(max_features, embed_size, weights=[final_embedding], trainable = False)(inp)\n    x = SpatialDropout1D(0.2)(x)\n    \n    x1 = Bidirectional(CuDNNLSTM(256, return_sequences=True))(x)\n    x1 = Bidirectional(CuDNNLSTM(128, return_sequences=True))(x1)\n    #x1 = Attention()(x1)\n    x1 = Conv1D(64, kernel_size=3, padding = \"valid\", kernel_initializer = \"he_uniform\")(x1)\n    \n    x2 = Bidirectional(CuDNNLSTM(256, return_sequences=True))(x)\n    x2 = Bidirectional(CuDNNGRU(128, return_sequences=True))(x2)\n    #x2 = Attention()(x2)\n    x2 = Conv1D(64, kernel_size=4, padding = \"valid\", kernel_initializer = \"he_uniform\")(x2)\n    \n    avg_pool1 = GlobalAveragePooling1D()(x1)\n    max_pool1 = GlobalMaxPooling1D()(x1)\n    \n    avg_pool2 = GlobalAveragePooling1D()(x2)\n    max_pool2 = GlobalMaxPooling1D()(x2)\n    \n    conc = concatenate([avg_pool1, max_pool1,avg_pool2, max_pool2])\n    conc = Dense(128, activation='relu')(conc)\n    conc = Dense(64, activation='relu')(conc)\n    conc = Dropout(0.1)(conc)\n    outp = Dense(1, activation='sigmoid')(conc)\n    model = Model(inputs = inp, outputs = outp)\n    model.compile(loss='binary_crossentropy', optimizer=Adam(lr = 1e-3, decay=0.0), metrics=['accuracy'])\n    return model","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"model = model_lstm_du(final_matrix)\nmodel.summary()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"gc.collect()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def train_pred(model, epoch=2):\n    filepath=\"weights_best.h5\"\n    checkpoint = ModelCheckpoint(filepath=filepath, monitor='val_loss', verbose=2, save_best_only=True, mode='min' )\n    #reduce_lr = ReduceLROnPlateau(monitor='val_loss', factor=0.0001, patience=1, verbose=2, min_lr=0.0001)\n    early_stopping = EarlyStopping(monitor='val_loss', min_delta=0.0001, verbose=1,patience=1, mode='min')\n    callbacks = [checkpoint,early_stopping]#, reduce_lr]\n    \n    #for e in range(epoch):\n    model.fit(x=X_train,y=Y_train, batch_size=512,epochs=epoch, callbacks=callbacks, validation_data=(X_val, Y_val))\n    model.load_weights(filepath)\n    pred_val_y = model.predict([X_val], batch_size=1024, verbose=0)\n    pred_test_y = model.predict([X_test], batch_size=1024, verbose=0)\n    return pred_val_y, pred_test_y","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"pred_val_y, pred_test_y = train_pred(model, epoch=8)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"submission = pd.read_csv(\"../input/jigsaw-unintended-bias-in-toxicity-classification/sample_submission.csv\")\nsubmission['prediction'] = pred_test_y\nsubmission.to_csv('submission.csv', index=False)","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.4","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}