{"cells":[{"metadata":{"_uuid":"03d5a6f8ebb86021e0893f4eaa7a7c8953c0024e"},"cell_type":"markdown","source":"**import**"},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport tensorflow as tf\n\nfrom sklearn.model_selection import train_test_split\nfrom sklearn import metrics\n\nfrom keras.preprocessing.text import Tokenizer\nfrom keras.preprocessing.sequence import pad_sequences\n\nimport matplotlib.pyplot as plt\n\nimport random\nimport copy\nimport time\nimport pandas as pd\nimport numpy as np\nimport gc\nimport re\nimport os \n\nfrom collections import Counter\nfrom nltk import word_tokenize\n\n# cross validation and metrics\nfrom sklearn.model_selection import StratifiedKFold, KFold\nfrom sklearn.metrics import *\n\nfrom sklearn.preprocessing import StandardScaler\nfrom multiprocessing import  Pool\nfrom functools import partial\nimport numpy as np\nfrom sklearn.decomposition import PCA\nimport keras\n","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"0c5b303785215b7e5231a937a107e96e19699765"},"cell_type":"markdown","source":"**config**"},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"embed_size = 300 # how big is each word vector\nmax_features = 120000 # how many unique words to use (i.e num rows in embedding vector)\nmaxlen = 70 # max number of words in a question to use\nbatch_size = 512 # how many samples to process at once\nn_epochs = 5 # how many times to iterate over all samples\nn_splits = 5 # Number of K-fold Splits\nSEED = 10\ndebug =0","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"8b89f32c79d42db2cbd169c84b33bfc059ad26f9"},"cell_type":"markdown","source":"**load embeddings**"},{"metadata":{"trusted":true,"_uuid":"4981f7278620d7fbe3aa6c4fcf31aee814b861c5"},"cell_type":"code","source":"def load_glove(word_index):\n    EMBEDDING_FILE = '../input/embeddings/glove.840B.300d/glove.840B.300d.txt'\n    def get_coefs(word,*arr): return word, np.asarray(arr, dtype='float32')[:300]\n    embeddings_index = dict(get_coefs(*o.split(\" \")) for o in open(EMBEDDING_FILE))\n    \n    all_embs = np.stack(embeddings_index.values())\n    emb_mean,emb_std = -0.005838499,0.48782197\n    embed_size = all_embs.shape[1]\n\n    # word_index = tokenizer.word_index\n    nb_words = min(max_features, len(word_index))\n    embedding_matrix = np.random.normal(emb_mean, emb_std, (nb_words, embed_size))\n    for word, i in word_index.items():\n        if i >= max_features: continue\n        embedding_vector = embeddings_index.get(word)\n        #ALLmight\n        if embedding_vector is not None: \n            embedding_matrix[i] = embedding_vector\n        else:\n            embedding_vector = embeddings_index.get(word.capitalize())\n            if embedding_vector is not None: \n                embedding_matrix[i] = embedding_vector\n    return embedding_matrix \n    \n            \ndef load_fasttext(word_index):    \n    EMBEDDING_FILE = '../input/embeddings/wiki-news-300d-1M/wiki-news-300d-1M.vec'\n    def get_coefs(word,*arr): return word, np.asarray(arr, dtype='float32')\n    embeddings_index = dict(get_coefs(*o.split(\" \")) for o in open(EMBEDDING_FILE) if len(o)>100)\n\n    all_embs = np.stack(embeddings_index.values())\n    emb_mean,emb_std = all_embs.mean(), all_embs.std()\n    embed_size = all_embs.shape[1]\n\n    # word_index = tokenizer.word_index\n    nb_words = min(max_features, len(word_index))\n    embedding_matrix = np.random.normal(emb_mean, emb_std, (nb_words, embed_size))\n    for word, i in word_index.items():\n        if i >= max_features: continue\n        embedding_vector = embeddings_index.get(word)\n        if embedding_vector is not None: embedding_matrix[i] = embedding_vector\n    return embedding_matrix\n\ndef load_para(word_index):\n    EMBEDDING_FILE = '../input/embeddings/paragram_300_sl999/paragram_300_sl999.txt'\n    def get_coefs(word,*arr): return word, np.asarray(arr, dtype='float32')\n    embeddings_index = dict(get_coefs(*o.split(\" \")) for o in open(EMBEDDING_FILE, encoding=\"utf8\", errors='ignore') if len(o)>100)\n\n    all_embs = np.stack(embeddings_index.values())\n    emb_mean,emb_std = -0.0053247833,0.49346462\n    embed_size = all_embs.shape[1]\n\n    # word_index = tokenizer.word_index\n    nb_words = min(max_features, len(word_index))\n    embedding_matrix = np.random.normal(emb_mean, emb_std, (nb_words, embed_size))\n    for word, i in word_index.items():\n        if i >= max_features: continue\n        embedding_vector = embeddings_index.get(word)\n        if embedding_vector is not None: embedding_matrix[i] = embedding_vector    \n    return embedding_matrix","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"8538f05474d5fe00facc329e2c27448cf5e9dc50"},"cell_type":"markdown","source":"**clean data and add features**"},{"metadata":{"trusted":true,"_uuid":"be9abffeefd68c3443eef624fa2ed404a1424ad8"},"cell_type":"code","source":"puncts = [',', '.', '\"', ':', ')', '(', '-', '!', '?', '|', ';', \"'\", '$', '&', '/', '[', ']', '>', '%', '=', '#', '*', '+', '\\\\', '•',  '~', '@', '£', \n '·', '_', '{', '}', '©', '^', '®', '`',  '<', '→', '°', '€', '™', '›',  '♥', '←', '×', '§', '″', '′', 'Â', '█', '½', 'à', '…', \n '“', '★', '”', '–', '●', 'â', '►', '−', '¢', '²', '¬', '░', '¶', '↑', '±', '¿', '▾', '═', '¦', '║', '―', '¥', '▓', '—', '‹', '─', \n '▒', '：', '¼', '⊕', '▼', '▪', '†', '■', '’', '▀', '¨', '▄', '♫', '☆', 'é', '¯', '♦', '¤', '▲', 'è', '¸', '¾', 'Ã', '⋅', '‘', '∞', \n '∙', '）', '↓', '、', '│', '（', '»', '，', '♪', '╩', '╚', '³', '・', '╦', '╣', '╔', '╗', '▬', '❤', 'ï', 'Ø', '¹', '≤', '‡', '√', ]\n\ndef clean_text(x):\n    x = str(x)\n    for punct in puncts:\n        if punct in x:\n            x = x.replace(punct, ' {punct} ')\n    return x\n\n\ndef clean_numbers(x):\n    if bool(re.search(r'\\d', x)):\n        x = re.sub('[0-9]{5,}', '#####', x)\n        x = re.sub('[0-9]{4}', '####', x)\n        x = re.sub('[0-9]{3}', '###', x)\n        x = re.sub('[0-9]{2}', '##', x)\n    return x\n\nmispell_dict = {\"ain't\": \"is not\", \"aren't\": \"are not\",\"can't\": \"cannot\", \"'cause\": \"because\", \"could've\": \"could have\", \"couldn't\": \"could not\", \"didn't\": \"did not\",  \"doesn't\": \"does not\", \"don't\": \"do not\", \"hadn't\": \"had not\", \"hasn't\": \"has not\", \"haven't\": \"have not\", \"he'd\": \"he would\",\"he'll\": \"he will\", \"he's\": \"he is\", \"how'd\": \"how did\", \"how'd'y\": \"how do you\", \"how'll\": \"how will\", \"how's\": \"how is\",  \"I'd\": \"I would\", \"I'd've\": \"I would have\", \"I'll\": \"I will\", \"I'll've\": \"I will have\",\"I'm\": \"I am\", \"I've\": \"I have\", \"i'd\": \"i would\", \"i'd've\": \"i would have\", \"i'll\": \"i will\",  \"i'll've\": \"i will have\",\"i'm\": \"i am\", \"i've\": \"i have\", \"isn't\": \"is not\", \"it'd\": \"it would\", \"it'd've\": \"it would have\", \"it'll\": \"it will\", \"it'll've\": \"it will have\",\"it's\": \"it is\", \"let's\": \"let us\", \"ma'am\": \"madam\", \"mayn't\": \"may not\", \"might've\": \"might have\",\"mightn't\": \"might not\",\"mightn't've\": \"might not have\", \"must've\": \"must have\", \"mustn't\": \"must not\", \"mustn't've\": \"must not have\", \"needn't\": \"need not\", \"needn't've\": \"need not have\",\"o'clock\": \"of the clock\", \"oughtn't\": \"ought not\", \"oughtn't've\": \"ought not have\", \"shan't\": \"shall not\", \"sha'n't\": \"shall not\", \"shan't've\": \"shall not have\", \"she'd\": \"she would\", \"she'd've\": \"she would have\", \"she'll\": \"she will\", \"she'll've\": \"she will have\", \"she's\": \"she is\", \"should've\": \"should have\", \"shouldn't\": \"should not\", \"shouldn't've\": \"should not have\", \"so've\": \"so have\",\"so's\": \"so as\", \"this's\": \"this is\",\"that'd\": \"that would\", \"that'd've\": \"that would have\", \"that's\": \"that is\", \"there'd\": \"there would\", \"there'd've\": \"there would have\", \"there's\": \"there is\", \"here's\": \"here is\",\"they'd\": \"they would\", \"they'd've\": \"they would have\", \"they'll\": \"they will\", \"they'll've\": \"they will have\", \"they're\": \"they are\", \"they've\": \"they have\", \"to've\": \"to have\", \"wasn't\": \"was not\", \"we'd\": \"we would\", \"we'd've\": \"we would have\", \"we'll\": \"we will\", \"we'll've\": \"we will have\", \"we're\": \"we are\", \"we've\": \"we have\", \"weren't\": \"were not\", \"what'll\": \"what will\", \"what'll've\": \"what will have\", \"what're\": \"what are\",  \"what's\": \"what is\", \"what've\": \"what have\", \"when's\": \"when is\", \"when've\": \"when have\", \"where'd\": \"where did\", \"where's\": \"where is\", \"where've\": \"where have\", \"who'll\": \"who will\", \"who'll've\": \"who will have\", \"who's\": \"who is\", \"who've\": \"who have\", \"why's\": \"why is\", \"why've\": \"why have\", \"will've\": \"will have\", \"won't\": \"will not\", \"won't've\": \"will not have\", \"would've\": \"would have\", \"wouldn't\": \"would not\", \"wouldn't've\": \"would not have\", \"y'all\": \"you all\", \"y'all'd\": \"you all would\",\"y'all'd've\": \"you all would have\",\"y'all're\": \"you all are\",\"y'all've\": \"you all have\",\"you'd\": \"you would\", \"you'd've\": \"you would have\", \"you'll\": \"you will\", \"you'll've\": \"you will have\", \"you're\": \"you are\", \"you've\": \"you have\", 'colour': 'color', 'centre': 'center', 'favourite': 'favorite', 'travelling': 'traveling', 'counselling': 'counseling', 'theatre': 'theater', 'cancelled': 'canceled', 'labour': 'labor', 'organisation': 'organization', 'wwii': 'world war 2', 'citicise': 'criticize', 'youtu ': 'youtube ', 'Qoura': 'Quora', 'sallary': 'salary', 'Whta': 'What', 'narcisist': 'narcissist', 'howdo': 'how do', 'whatare': 'what are', 'howcan': 'how can', 'howmuch': 'how much', 'howmany': 'how many', 'whydo': 'why do', 'doI': 'do I', 'theBest': 'the best', 'howdoes': 'how does', 'mastrubation': 'masturbation', 'mastrubate': 'masturbate', \"mastrubating\": 'masturbating', 'pennis': 'penis', 'Etherium': 'Ethereum', 'narcissit': 'narcissist', 'bigdata': 'big data', '2k17': '2017', '2k18': '2018', 'qouta': 'quota', 'exboyfriend': 'ex boyfriend', 'airhostess': 'air hostess', \"whst\": 'what', 'watsapp': 'whatsapp', 'demonitisation': 'demonetization', 'demonitization': 'demonetization', 'demonetisation': 'demonetization'}\n\ndef _get_mispell(mispell_dict):\n    mispell_re = re.compile('(%s)' % '|'.join(mispell_dict.keys()))\n    return mispell_dict, mispell_re\n\nmispellings, mispellings_re = _get_mispell(mispell_dict)\ndef replace_typical_misspell(text):\n    def replace(match):\n        return mispellings[match.group(0)]\n    return mispellings_re.sub(replace, text)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e52a6f626c4d68e99d083973c83bdb0f18d8e93e"},"cell_type":"code","source":"def parallelize_apply(df,func,colname,num_process,newcolnames):\n    # takes as input a df and a function for one of the columns in df\n    pool =Pool(processes=num_process)\n    arraydata = pool.map(func,tqdm(df[colname].values))\n    pool.close()\n    \n    newdf = pd.DataFrame(arraydata,columns = newcolnames)\n    df = pd.concat([df,newdf],axis=1)\n    return df\n\ndef parallelize_dataframe(df, func):\n    df_split = np.array_split(df, 4)\n    pool = Pool(4)\n    df = pd.concat(pool.map(func, df_split))\n    pool.close()\n    pool.join()\n    return df\n\ndef count_regexp_occ(regexp=\"\", text=None):\n    \"\"\" Simple way to get the number of occurence of a regex\"\"\"\n    return len(re.findall(regexp, text))\n\n\n# some fetures \n# ['','','india/n','','quora','','sex','','','country/countries','china','','','chinese','','']\ndef add_features(df):\n    df['question_text'] = df['question_text'].apply(lambda x:str(x))\n    df[\"lower_question_text\"] = df[\"question_text\"].apply(lambda x: x.lower())\n    # df = parallelize_apply(df,sentiment,'question_text',4,['sentiment','subjectivity']) \n    # df['sentiment'] = df['question_text'].progress_apply(lambda x:sentiment(x))\n    df['total_length'] = df['question_text'].apply(len)\n    df['capitals'] = df['question_text'].apply(lambda comment: sum(1 for c in comment if c.isupper()))\n    df['caps_vs_length'] = df.apply(lambda row: float(row['capitals'])/float(row['total_length']),axis=1)\n    df['num_words'] = df.question_text.str.count('\\S+')\n    df['num_unique_words'] = df['question_text'].apply(lambda comment: len(set(w for w in comment.split())))\n    df['words_vs_unique'] = df['num_unique_words'] / df['num_words'] \n    return df\n\ndef load_and_prec():\n    if debug:\n        train_df = pd.read_csv(\"../input/train.csv\")[:80000]\n        test_df = pd.read_csv(\"../input/test.csv\")[:20000]\n    else:\n        train_df = pd.read_csv(\"../input/train.csv\")\n        test_df = pd.read_csv(\"../input/test.csv\")\n    print(\"Train shape : \",train_df.shape)\n    print(\"Test shape : \",test_df.shape)\n    \n    ###################### Add Features ###############################\n    #  https://github.com/wongchunghang/toxic-comment-challenge-lstm/blob/master/toxic_comment_9872_model.ipynb\n    \n    train = add_features(train_df)\n    test = add_features(test_df)\n    \n#     train = parallelize_dataframe(train_df, add_features)\n#     test = parallelize_dataframe(test_df, add_features)\n    \n    # lower\n    train_df[\"question_text\"] = train_df[\"question_text\"].apply(lambda x: x.lower())\n    test_df[\"question_text\"] = test_df[\"question_text\"].apply(lambda x: x.lower())\n\n    # Clean the text\n    train_df[\"question_text\"] = train_df[\"question_text\"].apply(lambda x: clean_text(x))\n    test_df[\"question_text\"] = test_df[\"question_text\"].apply(lambda x: clean_text(x))\n    \n    # Clean numbers\n    train_df[\"question_text\"] = train_df[\"question_text\"].apply(lambda x: clean_numbers(x))\n    test_df[\"question_text\"] = test_df[\"question_text\"].apply(lambda x: clean_numbers(x))\n    \n    # Clean speelings\n    train_df[\"question_text\"] = train_df[\"question_text\"].apply(lambda x: replace_typical_misspell(x))\n    test_df[\"question_text\"] = test_df[\"question_text\"].apply(lambda x: replace_typical_misspell(x))\n    \n    ## fill up the missing values\n    train_X = train_df[\"question_text\"].fillna(\"_##_\").values\n    test_X = test_df[\"question_text\"].fillna(\"_##_\").values\n\n\n    \n    features = train[['num_unique_words','words_vs_unique', 'total_length', 'capitals', 'caps_vs_length','num_words']].fillna(0)\n    test_features = test[['num_unique_words','words_vs_unique', 'total_length', 'capitals', 'caps_vs_length','num_words']].fillna(0)\n       \n    # doing PCA to reduce network run times\n    ss = StandardScaler()\n    pc = PCA(n_components=5)\n    ss.fit(np.vstack((features, test_features)))\n    features = ss.transform(features)\n    test_features = ss.transform(test_features)\n    print('features shape: ', features.shape)\n    \n    ###########################################################################\n\n    ## Tokenize the sentences\n    tokenizer = Tokenizer(num_words=max_features)\n    tokenizer.fit_on_texts(list(train_X))\n    train_X = tokenizer.texts_to_sequences(train_X)\n    test_X = tokenizer.texts_to_sequences(test_X)\n\n    ## Pad the sentences \n    train_X = pad_sequences(train_X, maxlen=maxlen)\n    test_X = pad_sequences(test_X, maxlen=maxlen)\n\n    ## Get the target values\n    train_y = train_df['target'].values\n    \n#     # Splitting to training and a final test set    \n#     train_X, x_test_f, train_y, y_test_f = train_test_split(list(zip(train_X,features)), train_y, test_size=0.2, random_state=SEED)    \n#     train_X, features = zip(*train_X)\n#     x_test_f, features_t = zip(*x_test_f)    \n    \n    #shuffling the data\n    np.random.seed(SEED)\n    trn_idx = np.random.permutation(len(train_X))\n\n    train_X = train_X[trn_idx]\n    train_y = train_y[trn_idx]\n    features = features[trn_idx]\n    \n    return train_X, test_X, train_y, features, test_features, tokenizer.word_index\n#     return train_X, test_X, train_y, x_test_f,y_test_f,features, test_features, features_t, tokenizer.word_index\n#     return train_X, test_X, train_y, tokenizer.word_index","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"14edbbdbda47aa5b1c6a0a88339de5398fa19a7c"},"cell_type":"code","source":"start = time.time()\n# fill up the missing values\n# x_train, x_test, y_train, word_index = load_and_prec()\nx_train, x_test, y_train, features, test_features, word_index = load_and_prec() \n# x_train, x_test, y_train, x_test_f,y_test_f,features, test_features,features_t, word_index = load_and_prec() \nprint(time.time()-start)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"bf81d3b77d96fafc372ccf6f28238373bc9348d7"},"cell_type":"markdown","source":"**load embeddings**"},{"metadata":{"trusted":true,"_uuid":"336f74a924029419e2997a5bc59616a2594316df"},"cell_type":"code","source":"if debug:\n    paragram_embeddings = np.random.randn(120000,300)\n    glove_embeddings = np.random.randn(120000,300)\n    embedding_matrix = np.mean([glove_embeddings, paragram_embeddings], axis=0)\nelse:\n    glove_embeddings = load_glove(word_index)\n    embedding_matrix = glove_embeddings\n#     paragram_embeddings = load_para(word_index)\n#     embedding_matrix = np.mean([glove_embeddings, paragram_embeddings], axis=0)    \nnp.shape(embedding_matrix)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"8d3b948a455fb955b1b006572918be0ae4c69c24"},"cell_type":"markdown","source":"**lstm+gru+add features model**"},{"metadata":{"trusted":true,"_uuid":"f0bbad531af29e6cff535435a23f4e6dc0df0ccc"},"cell_type":"code","source":"from keras.models import *\nfrom keras.layers import *\nfrom keras import backend as K \n\nfeature_size = features.shape[1]\nprint('features_size: ', feature_size, 'train_x: ', x_train.shape, 'maxlen: ', maxlen)\ndef get_model(hidden_size, lin_size, embedding_matrix=embedding_matrix):\n    sequence_input = Input(shape=(maxlen, ), dtype='int32')   \n    embedding_layer = Embedding(input_dim=len(embedding_matrix),output_dim=embed_size, weights=[embedding_matrix], \n                                input_length=maxlen,trainable=False)\n    embedded_sequences = embedding_layer(sequence_input)\n    \n    x = Bidirectional(CuDNNLSTM(hidden_size, return_sequences=True))(embedded_sequences)\n    x = Bidirectional(CuDNNGRU(hidden_size, return_sequences=True))(x)    \n\n    avg_pool = GlobalAveragePooling1D()(x)\n    max_pool = GlobalMaxPooling1D()(x)\n    \n    feature_input = Input(shape=(feature_size, )) \n#     conc = concatenate([avg_pool, max_pool, feature_input])\n    hidden_trans = Lambda(lambda x: K.permute_dimensions(x, (1,0,2))[-1])\n    hidden_outp = hidden_trans(x)\n    print(hidden_outp.shape)\n    conc = concatenate([hidden_outp, avg_pool, max_pool, feature_input], axis=1)\n    outp = Dense(lin_size, activation=\"relu\")(conc)\n    outp = Dense(1, activation='sigmoid')(outp)\n\n    model = Model(inputs=[sequence_input, feature_input], outputs=outp)\n    model.compile(loss='binary_crossentropy', optimizer='adam', metrics=['accuracy'])\n    \n    return model","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"f9e238818fbed45a87edff360ef739888c7688b7"},"cell_type":"markdown","source":"**train and test**"},{"metadata":{"trusted":true,"_uuid":"04410a5680cf5d83c60580b28280fb28ef98c681"},"cell_type":"code","source":"splits = 5\nkf = StratifiedKFold(n_splits=splits, shuffle=True, random_state=SEED)\n\nbatch_size = 512\nhidden_size = 70\nlin_size = 16\n\npreds = []\npreds_vals = []\ny_vals = []\nfold = 0\nacc_scores = 0\n# os.environ['CUDA_VISIBLE_DEVICES'] = \"0\"\n# GPU = [0, 1]\n# for i in GPU:\n#     with tf.device('/gpu:{0}'.format(i)):\n# with tf.device('/gpu: 0'):\nfor train_idx, val_idx in kf.split(x_train, y_train):\n    x_train_f = x_train[train_idx]\n    f_train_f = features[train_idx]\n    y_train_f = y_train[train_idx]\n    x_val_f = x_train[val_idx]\n    f_val_f = features[val_idx]\n    y_val_f = y_train[val_idx]\n\n    # Output batch loss every epoch\n    model = get_model(hidden_size, lin_size)\n    model.fit([x_train_f, f_train_f], y_train_f,\n              batch_size=batch_size,\n              epochs=n_epochs,\n              verbose = 1,\n              validation_data=([x_val_f, f_val_f], y_val_f))\n    preds_val = model.predict([x_val_f, f_val_f], batch_size=batch_size)\n    preds_vals.append(preds_val)\n    y_vals.append(y_val_f)\n\n    preds_test = model.predict([x_test, test_features])\n    preds.append(preds_test)\n\n    fold+=1\n#         acc_scores += accuracy_score(y_val_f, preds_val)\n#         print('Fold {}, ACC = {}'.format(fold, accuracy_score(y_val_f, preds_val)))       \n#     print(\"Cross Validation ACC = {}\".format(acc_scores/splits))\n","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"195aab87b9f0f54a3bc3f5a39893baeb3a6eaa72"},"cell_type":"markdown","source":"**add f1 threshold**"},{"metadata":{"trusted":true,"_uuid":"3e7b77208ae1665b4ebee9cd3904ad11355a550d"},"cell_type":"code","source":"# threshold rearch\nbest_thresh = 0\nbest_score = 0\nfor thresh in np.arange(0.1,0.501,0.01):\n    thresh = np.round(thresh, 2)\n    scores = 0.\n    for i in range(len(preds_vals)):\n        score = f1_score(y_vals[i], (preds_vals[i]>thresh).astype(int))\n        scores += score\n    print(\"F1 score at threshold {0} is {1}\".format(thresh, scores/len(preds_vals)))\n    if score > best_score:\n        best_thresh = thresh\n        best_score = score\nprint (\"F1 score at threshold {0} is the best: {1}\".format(best_thresh, best_score))","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"6b00b93c4fb2505aee3c9567fa7ec0a3fa1ba857"},"cell_type":"markdown","source":"**submit**"},{"metadata":{"trusted":true,"_uuid":"c0dc6480f37d59ce5bcc394685292b71f58c83ef"},"cell_type":"code","source":"preds = np.asarray(preds)\nprint(preds.shape)\ny_test = np.mean(preds, axis=0)[:, 0]\nprint(y_test.shape)\ny_test = (y_test > 0.33).astype(np.int)\nsubmit_df = pd.DataFrame({\"qid\": test[\"qid\"], \"prediction\": y_test})\nsubmit_df.to_csv(\"submission_lstm_gru_addfeatures.csv\", index=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"3665fe6456c8a6b40bcaab68c508ad2676f65283"},"cell_type":"code","source":"import pandas as pd\nsubmit_df = pd.read_csv('../input/submissions/submission_lstm_gru_addfeatures.csv')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"4bd2c859c050b3558be612bb2f5dd6f9b13a0b97"},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}