{"cells":[{"metadata":{"trusted":true,"_uuid":"00ed96f5bf93d95fc2e7b47135653cf02c7f5b5d"},"cell_type":"code","source":"from os import listdir\nfrom os.path import isfile, join\nimport math\nimport re\nimport random\nimport time\nimport numpy as np\nimport pandas as pd\nfrom gensim.models import KeyedVectors\nfrom keras.models import Sequential, Model, load_model\nfrom keras.layers import Dense, GlobalAveragePooling1D, CuDNNGRU, Embedding, Bidirectional, BatchNormalization\nfrom keras.layers import Input, CuDNNLSTM, Lambda, concatenate, SpatialDropout1D, Dropout, GlobalMaxPooling1D\nfrom keras.preprocessing.text import Tokenizer\nfrom keras.preprocessing.sequence import pad_sequences\nfrom keras.callbacks import Callback\nfrom keras import backend as K\nfrom keras.engine.topology import Layer\nfrom keras import initializers, regularizers, constraints, optimizers, layers\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import f1_score, precision_score, recall_score\nfrom sklearn.svm import SVC","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"bdc09d06f795014016a1ea293b43f66ca3cd1df2"},"cell_type":"code","source":"num_features = 300\nnum_steps = 68\nbound_50 = 12\nbound_75 = 17\nbatch_size = 512\nclass_weight = {0:1, 1:15}\nval_size_percent = 0\nglove_epochs = 5\nwiki_epochs = 4\nnp.random.seed(seed=2019)\nnon_word_re = re.compile(\"([^a-zA-Z0-9'_])\")\nword_re = re.compile(\"([a-zA-Z0-9])\")\nnumber_re = re.compile(\"^\\d+(|k|s|p|m|'\\d+)$\")\norder_re = re.compile(\"^\\d+(th|1st|2nd|3rd)$\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"1fa5061ce02478e703da26ec69082da8eb97b471"},"cell_type":"code","source":"abbr_dict = {'doesnt': 'does not', 'didnt': 'did not', 'isnt': 'is not', 'masterbate': 'masturbate',\n             \"you've\": 'you have', \"isn't\": 'is not', \"aren't\": 'are not', \"won't\": 'will not', \n             \"Isn't\": 'Is not', \"they're\": 'they are', \"haven't\": 'have not', \"shouldn't\": 'should not', \n             \"hasn't\": 'has not', \"wasn't\": 'was not', \"couldn't\": 'could not', \"wouldn't\": 'would not', \n             \"Shouldn't\": 'Should not', \"Wouldn't\": 'Would not', \"Aren't\": 'Are not', \"we're\": 'we are', \n             \"weren't\": 'were not', \"they've\": 'they have', \"you'd\": 'you had', \"hadn't\": 'had not', \n             \"We're\": 'We are', \"we've\": 'we have', \"would've\": 'would have', \"Wasn't\": 'Was not', \n             \"they'll\": 'they will', \"he'll\": 'he will', \"We've\": 'We have', \"Won't\": 'Will not', \n             \"They're\": 'They are', \"Couldn't\": 'Could not', \"they'd\": 'they had', \"it'll\": 'it will', \n             \"he'd\": 'he had', \"could've\": 'could have', \"ain't\": \"have not\", \"we'll\": 'we will', \n             \"You've\": 'You have', \"who've\": 'who have', \"don't\": 'do not', \"I'm\": 'I am', \n             \"What's\": 'What is', \"can't\": 'cannot', \"doesn't\": 'does not', \"it's\": 'it is', \n             \"didn't\": 'did not', \"I've\": 'I have', \"you're\": 'you are', \"It's\": 'It is', \n             \"what's\": 'what is', \"he's\": 'he is', \"that's\": 'that is', \"Don't\": 'Do not', \n             \"there's\": 'there is', \"she's\": 'she is', \"who's\": 'who is', \"I'll\": 'I will', \n             \"Who's\": 'Who is', \"I'd\": 'I would', \"Doesn't\": 'Does not', \"How's\": 'How is', \n             \"Can't\": 'Cannot', \"There's\": 'There is', \"He's\": 'He is', \"Where's\": 'Where is', \n             \"you'll\": 'you will', \"let's\": 'let us', \"You're\": 'You are', \"She's\": 'She is',\n             \"Didn't\": 'Did not', \"Let's\": 'Let us', \"else's\": 'others', \"other's\": 'others', \n             \"That's\": 'That is', \"What're\": 'What are', \"y'all\": 'you all', \"she'll\": 'she will', \n             \"Haven't\": 'Have not', \"Howcan\": 'How can', \"Howmuch\": 'How much', \"should've\": 'should have', \n             \"we'd\": 'We would', \"Hasn't\": 'Has not', \"Who'd\": 'Who would', \"that'll\": 'that will', \n             \"who'd\": 'who would', \"They've\": 'They have', \"she'd\": 'she had', \"Weren't\": 'Were not',\n             \"They'd\": \"They had\", \"We'll\": \"We will\"}","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"38ead8dbdf463c15c7a9242fc7facc941d603aef"},"cell_type":"code","source":"syn_dict = {\"narcist\": \"overly self-involved\", \"Quoras\": \"Quora\", \"ciswomen\": \"cis women\", \n            \"Narikoravar\": \"community\", \"hiraba\": \"killing noncombatants\", \"eyerollingly\": \"eye-rollingly\",\n            \"transgendering\": \"trans gendering\", \"Islamophilia\": \"Islam psychiatric disorder\",\n            \"cuckservative\": \"cuck servative\", \"IQers\": \"IQ people\", \"BringBackOurGirls\": \"Bring Back Our Girls\",\n            \"rituels\": \"rituel\", \"libtarded\": \"libtard\", \"dumbwit\": \"stupid person\", \"half-boglin\": \"half-human\",\n            \"boooooooored\": \"very bored\", \"queriers\": \"querier\", \"Turkist\": \"Turkish\", \"mehfils\": \"mehfil\",\n            \"sapiosexuals\": \"Sapio sexuals\", \"richless\": \"poor\", \"jewprofits\": \"jew profits\", \n            \"highcourt\": \"high court\", \"offendable\": \"offend able\", \"oversecularism\": \"over secularism\"}","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"cbb1b1c435d702fb186dd8042ef6851bb46ee2c2"},"cell_type":"code","source":"mis_dict = {\"savegely\": \"savagely\", \"Ra-apist\": \"rapist\", \"Khazari\": \"Khazaria\", \"F'king\": \"Fucking\",\n            \"vagena\": \"vagina\", \"rohingyan\": \"Rohingya\", \"feminisam\": \"feminism\", \"faantasy\": \"fantasy\",\n            \"inapropriately\": \"inappropriately\", \"thatwhy\": \"why\", \"ILEGALY\": \"ILLEGALLY\",\n            \"sychopantic\": \"sycophantic\", \"pansexuals\": \"pansexual\", \"immuture\": \"immature\",\n            \"europion\": \"european\", \"Whydid\": \"Why did\", \"ociopath\": \"sociopath\", \"womenizer\": \"womanizer\",\n            \"discreminate\": \"discriminate\", \"tamilzan\": \"tamilian\", \"unrulier\": \"more unruly\",\n            \"abnoxiously\": \"obnoxiously\", \"Bhangis\": \"Bhangi\", \"reasorces\": \"resources\",\n# \"nchausen\": \"Münchausen\", \"nmeh\": \"Dönmeh\", \"Ljub\": \"Ljubčo\",\n            \"relarionship\": \"relationship\", \"langague\": \"language\", \"zealote\": \"zealot\",\n            \"douchebaginess\": \"douchebag\", \"Turkeyball\": \"Turkish people\", \"GeorgiaBall\": \"Georgian people\",\n            \"matherfuckers\": \"mather fuckers\", \"cetrizine\": \"cetirizine\", \"demorcratic\": \"democratic\",\n            \"indoctrine\": \"indoctrinate\", \"Ameeican\": \"American\", \"motherflipping\": \"mother flipping\",\n            \"statsmam\": \"statesman\", \"ckrcumcision\": \"circumcision\", \"microagressions\": \"micro agressions\",\n            \"auto-depresive\": \"auto depressive\", \"centimiters\": \"centimeters\", \"xender\": \"xander\",\n            \"demcoratic\": \"democratic\", \"westbengal\": \"West Bengal\", \"culpablefor\": \"culpable for\",\n            \"underwares\": \"underwears\", \"castated\": \"castrated\", \"stroneger\": \"stronger\", \n            \"deadbody\": \"dead body\", \"climeb\": \"climb\", \"uruguaios\": \"Uruguayans\", \"Skkim\": \"Sikkim\",\n            \"gipsos\": \"gypsies\", \"masturbatikn\": \"masturbation\", \"Rohignyas\": \"Rohingyas\", \n            \"warmism\": \"warming\", \"simallar\": \"similar\", \"at9yers\": \"at 9 years\", \"unoin\": \"union\",\n            \"non-miniorites\": \"non-minorities\", \"immerson\": \"immersion\", \"ethnicites\": \"ethnicities\",\n            \"driend\": \"friend\", \"incomfortable\": \"uncomfortable\", \"stupdity\": \"stupidity\",\n            \"postitution\": \"prostitution\", \"gutlless\": \"gutless\"}","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"857ff37cfcdd2c4ebc77c63db813cb4a66ba4597"},"cell_type":"code","source":"def replace_word(sentence):\n    s = []\n    for word in sentence:\n        if word in abbr_dict:\n            s.extend(abbr_dict[word].split())\n        elif word in syn_dict:\n            s.extend(syn_dict[word].split())\n        elif word in mis_dict:\n            s.extend(mis_dict[word].split())\n        else:\n            s.append(word)\n    return s","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e03651cd50a2988d967a361f0aeaf0dda4f28fac"},"cell_type":"code","source":"def pre_process_text(df):\n    replaced_sentences = []\n    num_words = []\n    tot_words = []\n    for s in df.question_text:\n        r = non_word_re.sub(r' \\1 ', s).split()\n        r = replace_word(r)\n        t = len(r)\n        tot_words.append(t)\n        num_words.append(sum(1 for w in r if word_re.search(w)))\n        replaced_sentences.append(r)\n    df['question_text'] = replaced_sentences\n    df['num_words'] = num_words\n    df['tot_words'] = tot_words\n    return df","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d5701bd5acb54209096a5cd765a1ac7be1ef7dc7"},"cell_type":"code","source":"def text_to_sequence(tokenizer, df):\n    X = tokenizer.texts_to_sequences(df.question_text)\n    X = pad_sequences(X, maxlen=num_steps, padding='post')\n    df['question_text'] = list(X)\n    return df","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"9f8b2854c16a8bb8f3000055844166fe10825ce7"},"cell_type":"code","source":"def load_data():\n    print(\"Loading csv files ...\")\n    train_df = pd.read_csv(\"../input/train.csv\")\n    test_df = pd.read_csv(\"../input/test.csv\")\n    \n    print(\"Pre-processing text ...\")\n    pre_process_text(train_df)\n    pre_process_text(test_df)\n    \n    print(\"Mapping text to sequence ...\")\n    tokenizer = Tokenizer(filters='', lower=False)\n    tokenizer.fit_on_texts(train_df.question_text)\n    tokenizer.fit_on_texts(test_df.question_text)\n    \n    train_df = train_df[train_df.num_words > 1]\n    if val_size_percent > 0:\n        train_df, test_df = train_test_split(train_df, test_size=val_size_percent/100)\n    text_to_sequence(tokenizer, train_df)\n    text_to_sequence(tokenizer, test_df)    \n    print(f\"Train size: {len(train_df)}, Test size: {len(test_df)}\")\n    \n    return train_df, test_df, tokenizer.word_index","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"1a4e16417d0e0259422a7acd31f011c53e64381e"},"cell_type":"code","source":"train_df, test_df, word_index = load_data()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"572a8e8266954b8ad6a0b5e0fb902363c890c68d"},"cell_type":"code","source":"train_df.describe()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"285be831ea391ed9e4161b8b860d47635e9875f4"},"cell_type":"code","source":"train_1_df = train_df[train_df.tot_words <= bound_75]\ntrain_1_df.describe()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"71720303c29d0aca9adf182eaa6f497026e65b76"},"cell_type":"code","source":"train_2_df = train_df[train_df.tot_words > bound_75]\ntrain_2_df.describe()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"98cecaa6375f4943366a661c4dccfd82cbe628a3"},"cell_type":"code","source":"test_df.describe()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"6208f35b013fefacc99f0f388cd0a76aa05a477d"},"cell_type":"code","source":"test_0_df = test_df[test_df.num_words <= 1]\ntest_0_df.describe()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ff90c5590b87b46291765127df917c127a670480"},"cell_type":"code","source":"test_1_df = test_df[(test_df.num_words > 1) & (test_df.tot_words <= bound_75)]\ntest_1_df.describe()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"7ccaa87971db37ad2e5c27b1a6f6fef304f4a18f"},"cell_type":"code","source":"test_2_df = test_df[(test_df.num_words > 1) & (test_df.tot_words > bound_75)]\ntest_2_df.describe()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e1bd63854c23edd71e6f4747b42178f22b07e29b"},"cell_type":"code","source":"def get_coefs(word, *arr): \n    return word, np.asarray(arr, dtype='float32')\n\ndef load_embedding(path):\n    vocab = dict(get_coefs(*o.split(\" \")) for o in open(path, encoding=\"utf8\", errors='ignore') if len(o)>100)\n    return vocab, vocab\n\ndef load_glove():\n    print(\"Loading glove ...\")\n    return load_embedding('../input/embeddings/glove.840B.300d/glove.840B.300d.txt')\n\ndef load_wiki():\n    print(\"Loading wiki ...\")\n    return load_embedding('../input/embeddings/wiki-news-300d-1M/wiki-news-300d-1M.vec')\n\ndef load_paragram():\n    print(\"Loading paragram ...\")\n    return load_embedding('../input/embeddings/paragram_300_sl999/paragram_300_sl999.txt')\n\ndef load_word2vec():\n    print(\"Loading word2vec ...\")\n    path = '../input/embeddings/GoogleNews-vectors-negative300/GoogleNews-vectors-negative300.bin'\n    word2vec = KeyedVectors.load_word2vec_format(path, binary=True)\n    return word2vec.vocab, word2vec","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"db12271cea916d678ae57d131966bb01911916b7"},"cell_type":"code","source":"def parse_word_vec(word, vocab, vocab_dict):\n    if word in vocab:\n        return vocab_dict[word]\n    w = word.upper()\n    if w in vocab:\n        return vocab_dict[w]\n    w = word.lower()\n    if w in vocab:\n        return vocab_dict[w]\n    w = w.capitalize()\n    if w in vocab:\n        return vocab_dict[w]\n    if not word_re.search(word):\n        return vocab_dict['#']\n    if number_re.search(word):\n        return vocab_dict['9']\n    if order_re.search(word):\n        return vocab_dict['9th']\n    if word.startswith(\"'\") or word.endswith(\"'\"):\n        return parse_word_vec(word.strip(\"'\"), vocab, vocab_dict)\n    if word.endswith(\"'s\"):\n        return parse_word_vec(word[:-2], vocab, vocab_dict)\n    if word.endswith(\"ies\"):\n        v = parse_word_vec(word[:-3] + 'y', vocab, vocab_dict)\n        if v is not None:\n            return v\n    if word.endswith(\"es\"):\n        v = parse_word_vec(word[:-2], vocab, vocab_dict)\n        if v is not None:\n            return v \n    if word.endswith(\"s\"):\n        return parse_word_vec(word[:-1], vocab, vocab_dict)\n    if word.endswith(\"ism\"):\n        return parse_word_vec(word[:-3], vocab, vocab_dict)\n    if \"'\" in word:\n        return parse_word_vec(word.replace(\"'\", \"\"), vocab, vocab_dict)\n    if \"_\" in word:\n        return parse_word_vec(word.replace(\"_\", \"\"), vocab, vocab_dict)\n    return None","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"fa4ef7c1cdcac54dc5e96560e462b3088fccc354"},"cell_type":"code","source":"def build_embedding_matrix(load_vocab_func):\n    vocab, vocab_dict = load_vocab_func()\n    all_embs = np.stack(list(vocab_dict.values()))\n    emb_mean, emb_std = all_embs.mean(), all_embs.std()\n    embedding_matrix = np.random.normal(emb_mean, emb_std, (len(word_index) + 1, num_features))\n    for word, i in word_index.items():\n        v = parse_word_vec(word, vocab, vocab_dict)\n        if v is not None:\n            embedding_matrix[i] = v\n    return embedding_matrix","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"53531042b80e4e06f86bf521e5b8f0ee016df879"},"cell_type":"code","source":"def build_embedding_layer(load_vocab_func):\n    embedding_matrix = build_embedding_matrix(load_vocab_func)\n    return Embedding(len(embedding_matrix), num_features, weights=[embedding_matrix], trainable=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"7e4a6f7551707386d8d749b4d6909dd4bec70874"},"cell_type":"code","source":"emb_glove = build_embedding_layer(load_glove)\nprint(\"Done\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"0a2e9a891b9c1cc8ab798788366ec0146680eccd"},"cell_type":"code","source":"emb_wiki = build_embedding_layer(load_wiki)\nprint(\"Done\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"dc732ec91c09eacd7ac298756bc120236ada5dc4"},"cell_type":"code","source":"# emb_paragram = build_embedding_layer(load_paragram)\n# print(\"Done\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"fa915e5114876177ee789b91d2f3a0ee110dc3fc"},"cell_type":"code","source":"def build_model_branch(embedding, rnn):\n    rnn = Bidirectional(rnn)(embedding)\n    part = 2\n    cells = []\n    for i in range(part):\n        x = Lambda(lambda x: x[:,i::part,:])(rnn)\n        x = GlobalAveragePooling1D()(x)\n        x = Dropout(0.3)(x)\n        x = Dense(16, activation='relu')(x)\n        x = BatchNormalization()(x)\n        cells.append(x)\n    x = concatenate(cells)\n    x = Dropout(0.3)(x)\n    x = Dense(1, activation='sigmoid')(x)\n    return x","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"20018b7d50322e8bc876e339590605ccdf175e3e"},"cell_type":"code","source":"def build_rnn_model():\n    inp = Input(shape=(num_steps, num_features))\n    \n    outs = []\n    for rnn in [CuDNNGRU(200, return_sequences=True), CuDNNLSTM(200, return_sequences=True)]:\n        outs.append(build_model_branch(inp, rnn))\n    \n    x = concatenate(outs)\n    x = BatchNormalization()(x)\n    x = Dense(16, activation='relu')(x)\n    x = BatchNormalization()(x)\n    x = Dense(16, activation='relu')(x)\n    x = BatchNormalization()(x)\n    outp = Dense(1, activation='sigmoid')(x)\n    outs.append(outp)\n    return Model(inputs=inp, outputs=outs)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"6f06a050a7894647c7b7cbe5d58906f4f2402833"},"cell_type":"code","source":"rnn_model = build_rnn_model()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"0c9cc95a8c85b319d17c4460474a1a2fd41dd4a0"},"cell_type":"code","source":"def build_model(embedding_layer):\n    inp = Input(shape=(num_steps,))\n    emb = embedding_layer(inp)\n    x = SpatialDropout1D(0.3)(emb)\n    outp = rnn_model(x)\n    \n    mdl = Model(inputs=inp, outputs=outp)\n    mdl.compile(loss='binary_crossentropy', optimizer='adam', metrics=['accuracy'])\n    mdl.summary()\n    rnn_model.summary()\n    return mdl","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"89135a04f1ea0176ca91ef567e2194d31aadc602"},"cell_type":"code","source":"def get_pred_int(y_pred, rate):\n    y = y_pred.reshape(-1)\n    n = math.ceil(len(y) * rate)\n    idx = np.argsort(y)[-n]\n    threshold = y[idx]\n    return (y >= threshold).astype(int)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d99c30b110bc91ff813683872190f552a92e9516"},"cell_type":"code","source":"def find_rate(y_true, y_pred, min_rate, max_rate):\n    r_max, f1_max = 0, 0\n    y_max = np.zeros(len(y_true))\n    for i in range(min_rate * 10, max_rate * 10):\n        r = i * 0.001\n        y = get_pred_int(y_pred, r)\n        f1 = f1_score(y_true, y)\n        if f1 > f1_max:\n            f1_max = f1\n            r_max = r\n            y_max = y\n    return f1_max, r_max, y_max","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a73c0537a6695c5990cbb9203df28e4b3e57ebdf"},"cell_type":"code","source":"class F1Evaluation(Callback):\n    def __init__(self):\n        super(Callback, self).__init__()\n\n    def on_epoch_end(self, epoch, logs={}):\n        if val_size_percent > 0:\n            for desc, df, min_rate, max_rate in [(\"short texts\", test_1_df, 3, 7), (\"long texts\", test_2_df, 11, 15)]:\n                X = np.array(list(df.question_text))\n                y_preds = self.model.predict(X, batch_size=1024, verbose=0)\n                f1, rate, _ = find_rate(df.target.values, y_preds[-1], min_rate, max_rate)\n                print(f\"{desc} - f1 score: {f1}, positive rate: {rate}\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"971b55dc90e5b556b13302afe4f19619ce7e8743"},"cell_type":"code","source":"class LRScheduler(Callback):\n    def __init__(self):\n        super(Callback, self).__init__()\n        self.min_lr = 0.001\n        self.max_lr = 0.003\n        self.bound = 1200\n        self.step = (self.max_lr - self.min_lr) / self.bound\n        self.rate = 0.98\n        \n    def on_epoch_begin(self, epoch, logs=None):\n        K.set_value(self.model.optimizer.lr, self.min_lr)\n        \n    def on_batch_end(self, batch, logs=None):\n        lr = K.get_value(self.model.optimizer.lr)\n        if (val_size_percent > 0) and (batch % 100 == 0):\n            print(f\"\\n{batch}: {lr}\")\n        if batch < self.bound:\n            lr += self.step\n        elif batch <= 2 * self.bound:\n            lr -= self.step\n        else:\n            lr *= self.rate\n        K.set_value(self.model.optimizer.lr, lr)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a563313fcf51fa91cad18a3e90a82c0610d44d5d"},"cell_type":"code","source":"model = build_model(emb_glove)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"70d2c5f9b1eff054ab9919775e4ed7f4da8dd865"},"cell_type":"code","source":"model.fit(np.array(list(train_df.question_text)), [train_df.target.values] * 3, batch_size=batch_size, \n          class_weight=class_weight, epochs=glove_epochs, callbacks=[LRScheduler(), F1Evaluation()])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"4da8477300f92c5422f92969a83b1029b64782f6"},"cell_type":"code","source":"y_preds = model.predict(np.array(list(test_df.question_text)), batch_size=1024, verbose=0)\ntest_df['glove'] = y_preds[-1]\ntest_df.describe()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"7c47cda735ce19ea95d0eee7260712ccd14677bf"},"cell_type":"code","source":"model = build_model(emb_wiki)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e888d2096f43e733f9abf360e40d6760a88b4349"},"cell_type":"code","source":"model.fit(np.array(list(train_df.question_text)), [train_df.target.values] * 3, batch_size=batch_size, \n          class_weight=class_weight, epochs=wiki_epochs, callbacks=[LRScheduler(), F1Evaluation()])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"afdfebfb4c1366124819406d601983999d38f860"},"cell_type":"code","source":"y_preds = model.predict(np.array(list(test_df.question_text)), batch_size=1024, verbose=0)\ntest_df['wiki'] = y_preds[-1]\ntest_df.describe()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"8b0ff5bd9a58441771b910496192c17872239414"},"cell_type":"code","source":"test_0_df = test_df[test_df.num_words <= 1]\ntest_1_df = test_df[(test_df.num_words > 1) & (test_df.tot_words <= bound_75)]\ntest_2_df = test_df[(test_df.num_words > 1) & (test_df.tot_words > bound_75)]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c4e32211f498796b4103fe963dbc6ecdc1704be1"},"cell_type":"code","source":"sub_df = pd.DataFrame(test_2_df)\ny = sub_df.glove.values + sub_df.wiki.values\nif val_size_percent > 0:\n    f1, rate, y = find_rate(sub_df.target.values, y, 11, 15)\n    print(f\"combine long - f1 score: {f1}, positive rate: {rate}\")\nelse:\n    y = get_pred_int(y, 0.143)\nsub_df['prediction'] = y\nsub_df.describe()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c2893158aa00bc3b4e033b340f5bca44bf2022c1"},"cell_type":"code","source":"sub_1_df = pd.DataFrame(test_1_df)\ny = sub_1_df.glove.values + sub_1_df.wiki.values\nif val_size_percent > 0:\n    f1, rate, y = find_rate(sub_1_df.target.values, y, 3, 6)\n    print(f\"short long - f1 score: {f1}, positive rate: {rate}\")\nelse:\n    y = get_pred_int(y, 0.044)\nsub_1_df['prediction'] = y\nsub_df = sub_df.append(sub_1_df)\nsub_df.describe()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"23497eb8920fc9d7558a1578488886e8875f1497"},"cell_type":"code","source":"if len(test_0_df) > 0:\n    sub_1_df = pd.DataFrame(test_0_df)\n    sub_1_df['prediction'] = 1\n    sub_df = sub_df.append(sub_1_df)\n    sub_df.describe()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f6d3f2575759ca5ce2d7571910617e567773ccea"},"cell_type":"code","source":"if val_size_percent > 0:\n    f1 = f1_score(sub_df.target.values, sub_df.prediction.values)\n    p = precision_score(sub_df.target.values, sub_df.prediction.values)\n    r = recall_score(sub_df.target.values, sub_df.prediction.values)\n    print(f\"final result - f1 score: {f1}, precision: {p}, recall: {r}\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"219f4ae8f4aca6fff65ebc0e94961a1042205f7a"},"cell_type":"code","source":"if val_size_percent > 0:\n    y = test_df.glove.values + test_df.wiki.values\n    f1, rate, y = find_rate(test_df.target.values, y, 2, 10)\n    print(f\"f1 score: {f1}, positive rate: {rate}\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"04cdbb048d0197098a49cfb48681b5505fecf696"},"cell_type":"code","source":"sub_df.to_csv(\"submission.csv\", columns=['qid', 'prediction'], index_label=False, index=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"baaa395d03b87f91c990e6f7ffc1b1c244450d34"},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}