{"cells":[{"metadata":{"trusted":true,"_uuid":"00ed96f5bf93d95fc2e7b47135653cf02c7f5b5d"},"cell_type":"code","source":"import math\nimport re\nimport random\nimport numpy as np\nimport pandas as pd\nfrom gensim.models import KeyedVectors\nfrom keras.models import Sequential, Model\nfrom keras.layers import Dense, CuDNNGRU, Dropout, Embedding, SpatialDropout1D, GlobalAveragePooling1D\nfrom keras.layers import GlobalMaxPooling1D, CuDNNLSTM, Input, concatenate, Lambda, CuDNNLSTM\nfrom keras.layers import Conv1D, Flatten, average, BatchNormalization, Bidirectional\nfrom keras.preprocessing.text import Tokenizer\nfrom keras.preprocessing.sequence import pad_sequences\nfrom keras.callbacks import Callback\nfrom keras import backend as K\nfrom keras.engine.topology import Layer\nfrom keras import initializers, regularizers, constraints, optimizers, layers\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import f1_score, precision_score, recall_score\nfrom sklearn.utils import shuffle\nfrom sklearn.svm import SVC","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d1477c708b7842e6f1d28e5c6bb13f64f13ff97b"},"cell_type":"code","source":"num_features = 300\nnum_steps = 68\nbound_steps = 17\nbatch_size = 512\nval_size_percent = 0\nepochs = 6\nshort_epochs = 10\nnp.random.seed(seed=2019)\nnon_word_re = re.compile(\"([^a-zA-Z0-9'_])\")\nword_re = re.compile(\"([a-zA-Z0-9])\")\nnumber_re = re.compile(\"^\\d+(|k|s|p|m|'\\d+)$\")\norder_re = re.compile(\"^\\d+(th|1st|2nd|3rd)$\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"9d090fd677ec5dc314c2127af98686a944f6272e"},"cell_type":"code","source":"abbr_dict = {'doesnt': 'does not', 'didnt': 'did not', 'isnt': 'is not', 'masterbate': 'masturbate',\n             \"you've\": 'you have', \"isn't\": 'is not', \"aren't\": 'are not', \"won't\": 'will not', \n             \"Isn't\": 'Is not', \"they're\": 'they are', \"haven't\": 'have not', \"shouldn't\": 'should not', \n             \"hasn't\": 'has not', \"wasn't\": 'was not', \"couldn't\": 'could not', \"wouldn't\": 'would not', \n             \"Shouldn't\": 'Should not', \"Wouldn't\": 'Would not', \"Aren't\": 'Are not', \"we're\": 'we are', \n             \"weren't\": 'were not', \"they've\": 'they have', \"you'd\": 'you had', \"hadn't\": 'had not', \n             \"We're\": 'We are', \"we've\": 'we have', \"would've\": 'would have', \"Wasn't\": 'Was not', \n             \"they'll\": 'they will', \"he'll\": 'he will', \"We've\": 'We have', \"Won't\": 'Will not', \n             \"They're\": 'They are', \"Couldn't\": 'Could not', \"they'd\": 'they had', \"it'll\": 'it will', \n             \"he'd\": 'he had', \"could've\": 'could have', \"ain't\": \"have not\", \"we'll\": 'we will', \n             \"You've\": 'You have', \"who've\": 'who have', \"don't\": 'do not', \"I'm\": 'I am', \n             \"What's\": 'What is', \"can't\": 'cannot', \"doesn't\": 'does not', \"it's\": 'it is', \n             \"didn't\": 'did not', \"I've\": 'I have', \"you're\": 'you are', \"It's\": 'It is', \n             \"what's\": 'what is', \"he's\": 'he is', \"that's\": 'that is', \"Don't\": 'Do not', \n             \"there's\": 'there is', \"she's\": 'she is', \"who's\": 'who is', \"I'll\": 'I will', \n             \"Who's\": 'Who is', \"I'd\": 'I would', \"Doesn't\": 'Does not', \"How's\": 'How is', \n             \"Can't\": 'Cannot', \"There's\": 'There is', \"He's\": 'He is', \"Where's\": 'Where is', \n             \"you'll\": 'you will', \"let's\": 'let us', \"You're\": 'You are', \"She's\": 'She is',\n             \"Didn't\": 'Did not', \"Let's\": 'Let us', \"else's\": 'others', \"other's\": 'others', \n             \"That's\": 'That is', \"What're\": 'What are', \"y'all\": 'you all', \"she'll\": 'she will', \n             \"Haven't\": 'Have not', \"Howcan\": 'How can', \"Howmuch\": 'How much', \"should've\": 'should have', \n             \"we'd\": 'We would', \"Hasn't\": 'Has not', \"Who'd\": 'Who would', \"that'll\": 'that will', \n             \"who'd\": 'who would', \"They've\": 'They have', \"she'd\": 'she had', \"Weren't\": 'Were not',\n             \"They'd\": \"They had\", \"We'll\": \"We will\"}","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"734a187360aa8709211cad86992131dbdb70887e"},"cell_type":"code","source":"syn_dict = {\"narcist\": \"overly self-involved\", \"Quoras\": \"Quora\", \"ciswomen\": \"cis women\", \n            \"Narikoravar\": \"community\", \"hiraba\": \"killing noncombatants\", \"eyerollingly\": \"eye-rollingly\",\n            \"transgendering\": \"trans gendering\", \"Islamophilia\": \"Islam psychiatric disorder\",\n            \"cuckservative\": \"cuck servative\", \"IQers\": \"IQ people\", \"BringBackOurGirls\": \"Bring Back Our Girls\",\n            \"rituels\": \"rituel\", \"libtarded\": \"libtard\", \"dumbwit\": \"stupid person\", \"half-boglin\": \"half-human\",\n            \"boooooooored\": \"very bored\", \"queriers\": \"querier\", \"Turkist\": \"Turkish\", \"mehfils\": \"mehfil\",\n            \"sapiosexuals\": \"Sapio sexuals\", \"richless\": \"poor\", \"jewprofits\": \"jew profits\", \n            \"highcourt\": \"high court\", \"offendable\": \"offend able\", \"oversecularism\": \"over secularism\"}","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"4b0e1f040390cfc2920f27fc1c4f960de2c0ffec"},"cell_type":"code","source":"mis_dict = {\"savegely\": \"savagely\", \"Ra-apist\": \"rapist\", \"Khazari\": \"Khazaria\", \"F'king\": \"Fucking\",\n            \"vagena\": \"vagina\", \"rohingyan\": \"Rohingya\", \"feminisam\": \"feminism\", \"faantasy\": \"fantasy\",\n            \"inapropriately\": \"inappropriately\", \"thatwhy\": \"why\", \"ILEGALY\": \"ILLEGALLY\",\n            \"sychopantic\": \"sycophantic\", \"pansexuals\": \"pansexual\", \"immuture\": \"immature\",\n            \"europion\": \"european\", \"Whydid\": \"Why did\", \"ociopath\": \"sociopath\", \"womenizer\": \"womanizer\",\n            \"discreminate\": \"discriminate\", \"tamilzan\": \"tamilian\", \"unrulier\": \"more unruly\",\n            \"abnoxiously\": \"obnoxiously\", \"Bhangis\": \"Bhangi\", \"reasorces\": \"resources\",\n# \"nchausen\": \"Münchausen\", \"nmeh\": \"Dönmeh\", \"Ljub\": \"Ljubčo\",\n            \"relarionship\": \"relationship\", \"langague\": \"language\", \"zealote\": \"zealot\",\n            \"douchebaginess\": \"douchebag\", \"Turkeyball\": \"Turkish people\", \"GeorgiaBall\": \"Georgian people\",\n            \"matherfuckers\": \"mather fuckers\", \"cetrizine\": \"cetirizine\", \"demorcratic\": \"democratic\",\n            \"indoctrine\": \"indoctrinate\", \"Ameeican\": \"American\", \"motherflipping\": \"mother flipping\",\n            \"statsmam\": \"statesman\", \"ckrcumcision\": \"circumcision\", \"microagressions\": \"micro agressions\",\n            \"auto-depresive\": \"auto depressive\", \"centimiters\": \"centimeters\", \"xender\": \"xander\",\n            \"demcoratic\": \"democratic\", \"westbengal\": \"West Bengal\", \"culpablefor\": \"culpable for\",\n            \"underwares\": \"underwears\", \"castated\": \"castrated\", \"stroneger\": \"stronger\", \n            \"deadbody\": \"dead body\", \"climeb\": \"climb\", \"uruguaios\": \"Uruguayans\", \"Skkim\": \"Sikkim\",\n            \"gipsos\": \"gypsies\", \"masturbatikn\": \"masturbation\", \"Rohignyas\": \"Rohingyas\", \n            \"warmism\": \"warming\", \"simallar\": \"similar\", \"at9yers\": \"at 9 years\", \"unoin\": \"union\",\n            \"non-miniorites\": \"non-minorities\", \"immerson\": \"immersion\", \"ethnicites\": \"ethnicities\",\n            \"driend\": \"friend\", \"incomfortable\": \"uncomfortable\", \"stupdity\": \"stupidity\",\n            \"postitution\": \"prostitution\", \"gutlless\": \"gutless\"}","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b121b30e5626e7449f23a209cc1ee73c5ae97f71"},"cell_type":"code","source":"def replace_word(sentence):\n    s = []\n    for word in sentence:\n        if word in abbr_dict:\n            s.extend(abbr_dict[word].split())\n        elif word in syn_dict:\n            s.extend(syn_dict[word].split())\n        elif word in mis_dict:\n            s.extend(mis_dict[word].split())\n        else:\n            s.append(word)\n    return s","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"955dea1e899e03002916fea16222b0024a22f31c"},"cell_type":"code","source":"def pre_process_text(df):\n    replaced_sentences = []\n    num_words = []\n    tot_words = []\n    for s in df.question_text:\n        r = non_word_re.sub(r' \\1 ', s).split()\n        r = replace_word(r)\n        t = len(r)\n        tot_words.append(t)\n        c = sum(1 for w in r if word_re.search(w))\n        num_words.append(c)\n        replaced_sentences.append(r)\n    df['question_text'] = replaced_sentences\n    df['num_words'] = num_words\n    df['tot_words'] = tot_words\n    return df","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"1b90663725f8179dc90077694459eca5d65083ac"},"cell_type":"code","source":"def text_to_sequence(tokenizer, df):\n    X = tokenizer.texts_to_sequences(df.question_text)\n    X = pad_sequences(X, maxlen=num_steps, padding='post')\n    df['question_text'] = list(X)\n    return df","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d3de21419639e2af294443b73b04ea90c19c1352"},"cell_type":"code","source":"def load_data():\n    print(\"Loading csv files ...\")\n    train_df = pd.read_csv(\"../input/train.csv\")\n    test_df = pd.read_csv(\"../input/test.csv\")\n    \n    print(\"Pre-processing text ...\")\n    pre_process_text(train_df)\n    pre_process_text(test_df)\n    \n    print(\"Mapping text to sequence ...\")\n    tokenizer = Tokenizer(filters='', lower=False)\n    tokenizer.fit_on_texts(train_df.question_text)\n    tokenizer.fit_on_texts(test_df.question_text)\n    \n    train_df = train_df[train_df.num_words > 1]\n    if val_size_percent > 0:\n        train_df, test_df = train_test_split(train_df, test_size=val_size_percent/100)\n    text_to_sequence(tokenizer, train_df)\n    text_to_sequence(tokenizer, test_df)    \n    print(f\"Train size: {len(train_df)}, Test size: {len(test_df)}\")\n    \n    return train_df, test_df, tokenizer.word_index","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a63a5cebf360380c2401358d990eaace1b037dab"},"cell_type":"code","source":"train_df, test_df, word_index = load_data()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b0e51ee20f3c447eec38d1c43b2adac64d318149"},"cell_type":"code","source":"train_df['rand'] = np.random.rand(len(train_df))\ntrain_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"9fc10a509a792f56b0c0b7f809c28b2a96e70a84"},"cell_type":"code","source":"test_df['rand'] = np.random.rand(len(test_df))\ntest_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"1c25b9bcf76302b3a913a0d186c37475bade8fd2"},"cell_type":"code","source":"test_0_df = test_df[test_df.num_words <= 1]\ntest_0_df.describe()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"9ecc6c19a022882e2007b39821882c79b9247993"},"cell_type":"code","source":"test_1_df = test_df[(test_df.num_words > 1) & (test_df.tot_words <= bound_steps)]\ntest_1_df.describe()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"9cda71fa9978853d15f23384fdf4b7d82b1cb276"},"cell_type":"code","source":"test_2_df = test_df[(test_df.num_words > 1) & (test_df.tot_words > bound_steps)]\ntest_2_df.describe()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f403ce5f6166562cb24351d721d64e98c793c16f"},"cell_type":"code","source":"def get_coefs(word, *arr): \n    return word, np.asarray(arr, dtype='float32')\n\ndef load_glove():\n    path = '../input/embeddings/glove.840B.300d/glove.840B.300d.txt'\n    glove = dict(get_coefs(*o.split(\" \")) for o in open(path) if len(o) > 100)\n    print(\"glove loaded\")\n    return glove","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"870806e602923ddbef29c026f36452cd1c268b6d"},"cell_type":"code","source":"def parse_word_vec(word, vocab):\n    if word in vocab:\n        return vocab[word]\n    w = word.upper()\n    if w in vocab:\n        return vocab[w]\n    w = word.lower()\n    if w in vocab:\n        return vocab[w]\n    w = w.capitalize()\n    if w in vocab:\n        return vocab[w]\n    if not word_re.search(word):\n        return vocab['#']\n    if number_re.search(word):\n        return vocab['9']\n    if order_re.search(word):\n        return vocab['9th']\n    if word.startswith(\"'\") or word.endswith(\"'\"):\n        return parse_word_vec(word.strip(\"'\"), vocab)\n    if word.endswith(\"'s\"):\n        return parse_word_vec(word[:-2], vocab)\n    if word.endswith(\"ies\"):\n        v = parse_word_vec(word[:-3] + 'y', vocab)\n        if v is not None:\n            return v\n    if word.endswith(\"es\"):\n        v = parse_word_vec(word[:-2], vocab)\n        if v is not None:\n            return v \n    if word.endswith(\"s\"):\n        return parse_word_vec(word[:-1], vocab)\n    if word.endswith(\"ism\"):\n        return parse_word_vec(word[:-3], vocab)\n    if \"'\" in word:\n        return parse_word_vec(word.replace(\"'\", \"\"), vocab)\n    if \"_\" in word:\n        return parse_word_vec(word.replace(\"_\", \"\"), vocab)\n    return None","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"0d4d781e65ec43e284046875c3f2e9779d1eefa2"},"cell_type":"code","source":"def build_embedding_matrix(load_vocab_func):\n    vocab = load_vocab_func()\n    all_embs = np.stack(list(vocab.values()))\n    emb_mean, emb_std = all_embs.mean(), all_embs.std()\n    embedding_matrix = np.random.normal(emb_mean, emb_std, (len(word_index) + 1, num_features))\n    for w, i in word_index.items():\n        v = parse_word_vec(w, vocab)\n        if v is not None:\n            embedding_matrix[i] = v\n    return embedding_matrix","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"dd8261dc9df5f24aa9b8a36debfad7a784ade5db"},"cell_type":"code","source":"embedding_matrix = build_embedding_matrix(load_glove)\nprint(\"Done\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"472fd67b9f3e1bcc68a9ee8c530017991cdc9288"},"cell_type":"code","source":"def build_model_branch(inp, rnn):\n    emb = Embedding(len(embedding_matrix), num_features, weights=[embedding_matrix], trainable=False)(inp)\n    x = SpatialDropout1D(0.3)(emb)\n    rnn = Bidirectional(rnn)(x)\n    part = 2\n    cells = []\n    for i in range(part):\n        x = Lambda(lambda x: x[:,i::part,:])(rnn)\n        x = GlobalAveragePooling1D()(x)\n        x = Dropout(0.3)(x)\n        x = Dense(16, activation='relu')(x)\n        cells.append(x)\n    conc = concatenate(cells)\n    x = Dropout(0.2)(conc)\n    x = Dense(1, activation='sigmoid')(x)\n    return x","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"aad854d140ee19baf7801f43d5c122ee2316402d"},"cell_type":"code","source":"def build_long_model():\n    inp = Input(shape=(num_steps,))\n    \n    u = build_model_branch(inp, CuDNNGRU(200, return_sequences=True))\n    v = build_model_branch(inp, CuDNNLSTM(200, return_sequences=True))\n    \n    x = concatenate([u, v])\n    x = Dense(8, activation='relu')(x)\n    x = Dropout(0.1)(x)\n    x = Dense(8, activation='relu')(x)\n    x = Dropout(0.1)(x)\n    outp = Dense(1, activation='sigmoid')(x)\n    \n    mdl = Model(inputs=inp, outputs=[outp, u, v])\n    mdl.compile(loss='binary_crossentropy', optimizer='adam', metrics=['accuracy'])\n    mdl.summary()\n    return mdl","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"736766bc64a4ac54c816ca32b3f845f0dc3fb906"},"cell_type":"code","source":"long_model = build_long_model()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"29914511ff7c8ab94ac3012f73401a5d8889e594"},"cell_type":"code","source":"def calculate_f1(y_true, y_pred, threshold):\n    y_pred_int = (y_pred > threshold).astype(int)\n    s = np.sum(y_pred_int)\n    rate = s / len(y_pred_int)\n    if (s == 0 or s == len(y_pred_int)):\n        return 0.0, y_pred_int, rate\n    f1 = f1_score(y_true, y_pred_int)\n    return f1, y_pred_int, rate","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"9ac1e22fd382f1696cc4d97fa00fb1f387578493"},"cell_type":"code","source":"def f1_matrics(y_true, y_pred):\n    t_max, rate_max, f1_max, p, r = 0.0, 0.0, 0.0, 0.0, 0.0\n    for i in range(200, 600, 5):\n        t = i * 0.001\n        f1, y_pred_int, rate = calculate_f1(y_true, y_pred, t)\n        if (f1 > f1_max):\n            f1_max, y_int_max, rate_max, t_max = f1, y_pred_int, rate, t\n    if f1_max > 0.0:\n        p = precision_score(y_true, y_int_max)\n        r = recall_score(y_true, y_int_max)\n    print(f\"[Threshold: {t_max:.6f}, Rate: {rate_max:.6f}, F1_Score: {f1_max:.6f}, Precision: {p:.6f}, Recall: {r:.6f}]\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"5cba5e801fc4daca6bb0e401910ad7a6e8072808"},"cell_type":"code","source":"class F1Evaluation(Callback):\n    def __init__(self, df):\n        super(Callback, self).__init__()\n        self.df = df\n\n    def on_epoch_end(self, epoch, logs={}):\n        if (val_size_percent > 0):\n            X = np.array(list(self.df.question_text))\n            y_preds = self.model.predict(X, batch_size=1024, verbose=0)\n            f1_matrics(self.df.target.values, y_preds[0])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f004f07fb4516152446f55717920d21dd2f06a45"},"cell_type":"code","source":"f1_eval = F1Evaluation(test_2_df)\nlong_model.fit(np.array(list(train_df.question_text)), [train_df.target.values] * 3, \n               batch_size=batch_size, epochs=epochs, callbacks=[f1_eval])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"38e8bd660d86459770a51a4b85ac95944875b6cd"},"cell_type":"code","source":"def get_pred_int(y_pred, rate):\n    y = y_pred.reshape(-1)\n    n = math.ceil(len(y) * rate)\n    idx = np.argsort(y)[-n]\n    threshold = y[idx]\n    return (y > threshold).astype(int), threshold","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"af673518be531e7719f80d187ebd0fd934d972a0"},"cell_type":"code","source":"y_preds = long_model.predict(np.array(list(test_2_df.question_text)), batch_size=1024, verbose=0)\nsub_df = pd.DataFrame(test_2_df)\nsub_df['prediction'], _ = get_pred_int(y_preds[0], 0.14)\nsub_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e6b1866827354b12b0a018c3f0d52c5336699f79"},"cell_type":"code","source":"y_preds = long_model.predict(np.array(list(test_1_df.question_text)), batch_size=1024, verbose=0)\nsub_1_df = pd.DataFrame(test_1_df)\nsub_1_df['prediction'], _ = get_pred_int(y_preds[0], 0.045)\nsub_1_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"77ed9abdb8813e7747714460fb7cae9836e2e321"},"cell_type":"code","source":"sub_df = sub_df.append(sub_1_df[sub_1_df.prediction == 1])\nsub_df.describe()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"97e5684d57a66770c03ba98594dcafdcd04a7921"},"cell_type":"code","source":"test_1_df = pd.DataFrame(sub_1_df[sub_1_df.prediction == 0])\ntest_1_df.describe()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"5d4ba632ed0d688503f0ef3028bd3a7e27406882"},"cell_type":"code","source":"def build_short_model():\n    inp = Input(shape=(bound_steps,))\n    emb = Embedding(len(embedding_matrix), num_features, weights=[embedding_matrix], trainable=False)(inp)\n    x = SpatialDropout1D(0.3)(emb)\n    x = Bidirectional(CuDNNLSTM(128, return_sequences=True), name=\"encoding\")(x)\n    x = Bidirectional(CuDNNLSTM(128, return_sequences=True))(x)\n    x = GlobalAveragePooling1D()(x)\n    x = Dropout(0.3)(x)\n    x = Dense(128, activation='relu')(x)\n    x = Dropout(0.3)(x)\n    x = Dense(128, activation='relu')(x)\n    x = Dropout(0.3)(x)\n    outp = Dense(1, kernel_regularizer=regularizers.l2(0.01))(x)\n    \n    mdl = Model(inputs=inp, outputs=outp)\n    mdl.compile(loss='hinge', optimizer='adam', metrics=['accuracy'])\n    mdl.summary()\n    return mdl","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"5e195655258cbe54d02b68f9170d0df71eaf03fb"},"cell_type":"code","source":"short_model = build_short_model()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"2759c9425b4f2c443e77a854643d72638c580ed7"},"cell_type":"code","source":"def hinge_f1_matrics(y_true, y_pred):\n    t_max, rate_max, f1_max, p, r = 0.0, 0.0, 0.0, 0.0, 0.0\n    for b in range(400, 10, -1):\n        y, t = get_pred_int(y_pred, b * 0.0001)\n        pt = precision_score(y_true, y)\n        if pt > p:\n            t_max = t\n            rate_max = np.sum(y) / len(y)\n            f1_max = f1_score(y_true, y)\n            p = pt\n            r = recall_score(y_true, y)\n            if pt > 0.7:\n                break\n    print(f\"[Threshold: {t_max:.6f}, Rate: {rate_max:.6f}, F1_Score: {f1_max:.6f}, Precision: {p:.6f}, Recall: {r:.6f}]\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"749be2b53ca53fa326ef1490ad4b6d48a2c9f7b2"},"cell_type":"code","source":"class HingeF1Evaluation(Callback):\n    def __init__(self):\n        super(Callback, self).__init__()\n        \n    def on_epoch_end(self, epoch, logs={}):\n         if val_size_percent > 0:\n            X = np.array(list(test_1_df.question_text))[:, :bound_steps]\n            y = self.model.predict(X, batch_size=1024, verbose=0)\n            hinge_f1_matrics(test_hinge_df.target.values, y)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"abb76f35d58a6d721395c29c8413d50e6250c33d"},"cell_type":"code","source":"train_1_df = train_df[train_df.tot_words <= bound_steps]\ntrain_1_df.describe()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"46d7e38fc611f26e6c0aa01e4af7ba12ba541815"},"cell_type":"code","source":"X = np.array(list(train_1_df.question_text))[:, :bound_steps]\ny = train_1_df.target.values * 2 - 1\nX, y = shuffle(X, y, random_state=2019)\nshort_model.fit(X, y, class_weight = {-1: 1, 1: 1}, batch_size=batch_size, epochs=short_epochs, callbacks=[HingeF1Evaluation()])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"877bf17a39520ee5c6074d41415e482d3344d398"},"cell_type":"code","source":"X = np.array(list(test_1_df.question_text))[:, :bound_steps]\ny_pred = short_model.predict(X, batch_size=1024, verbose=0)\nsub_1_df = pd.DataFrame(test_1_df)\nsub_1_df['prediction'], _ = get_pred_int(y_pred, 0.007)\nsub_1_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e2deee60e285fced3635b1d439228cc1bbfec862"},"cell_type":"code","source":"sub_df = sub_df.append(sub_1_df)\nsub_df.describe()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"10fe762302885bdd72cfd5e28b25ddb0af550798"},"cell_type":"code","source":"if len(test_0_df) > 0:\n    df = pd.DataFrame(test_0_df)\n    df['prediction'] = 1\n    sub_df.append(df)\nsub_df.to_csv(\"submission.csv\", columns=['qid', 'prediction'], index_label=False, index=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"1cb87ae493d1bfcdc11b1097cad9d588eae1b325"},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}