{"cells":[{"metadata":{"_uuid":"5ffd85f02755034cc83bc442cbd6a1f5488fddd8"},"cell_type":"markdown","source":"The purpose of this kernel is to study the effect of question lengths. There are many questions made up of multiple sentences. Most of those are insincere. When we pad (trim) the sequences to certain length, we may be trimming off the insincere component of an overall insincere question. This will eventually confuse the training and lead to misleading results (supposedly). To overcome this, I thought of feeding different parts of the sentence independently to the model for prediction and then either average all the predictions for that one sentence or do a weighted sum based on how confident the pred is. In this case, I used a small maxlen so that we still have decent number of questions over the maxlen limit. As it turns out, there aren't many. The results improved only incrementally. The trimming of lengthy sentences turned out to be a small issue. It might still help to feed length as a separate auxilliary input, but it does not pay off to process the remainder of the sentence.\n\nThis is my first public kernel, so please excuse the format. I would like to thank all the others who shared their fantastic kernels, especially Rahul Agarwal for sharing the basis for this one."},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import os\nimport json\nimport string\nimport numpy as np\nimport pandas as pd\nimport keras\nfrom pandas.io.json import json_normalize\nimport matplotlib.pyplot as plt\nimport seaborn as sns\ncolor = sns.color_palette()\nfrom math import floor\n\n%matplotlib inline\n\nfrom plotly import tools\nimport plotly.offline as py\npy.init_notebook_mode(connected=True)\nimport plotly.graph_objs as go\n\nfrom sklearn import model_selection, preprocessing, metrics, ensemble, naive_bayes, linear_model\nfrom sklearn.feature_extraction.text import TfidfVectorizer, CountVectorizer\nfrom sklearn.decomposition import TruncatedSVD\nimport lightgbm as lgb\n\nimport time\nfrom tqdm import tqdm\nimport math\nfrom sklearn.model_selection import train_test_split\n\nfrom keras.preprocessing.text import Tokenizer\nfrom keras.preprocessing.sequence import pad_sequences\nfrom keras.layers import Dense, Input, LSTM, Embedding, Dropout, Activation, GRU, CuDNNGRU, Conv1D\nfrom keras.layers import Bidirectional, GlobalMaxPool1D, Concatenate, Add, Flatten, CuDNNLSTM\nfrom keras.models import Model\nfrom keras import backend as K\nfrom keras import initializers, regularizers, constraints, optimizers, layers\nfrom keras.engine.topology import Layer\nfrom keras.callbacks import ModelCheckpoint, ReduceLROnPlateau, EarlyStopping\n\npd.options.mode.chained_assignment = None\npd.options.display.max_columns = 999\n\nfrom wordcloud import WordCloud, STOPWORDS\nfrom collections import defaultdict","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ffafdd1f01a4076c338e0de8b041f46ea84d8867"},"cell_type":"code","source":"import nltk \nfrom nltk.corpus import stopwords, wordnet\nfrom nltk.tokenize import word_tokenize, sent_tokenize \nstop_words = set(stopwords.words('english')) \nimport regex as re","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"3334562a999eb14eafa19a5e2fb5642e27deff6f"},"cell_type":"code","source":"puncts = [',', '.', '\"', ':', ')', '(', '-', '!', '?', '|', ';', \"'\", '$', '&', '/', '[', ']', '>', '%', '=', '#', '*', '+', '\\\\', '•',  '~', '@', '£', \n '·', '_', '{', '}', '©', '^', '®', '`',  '<', '→', '°', '€', '™', '›',  '♥', '←', '×', '§', '″', '′', 'Â', '█', '½', 'à', '…', \n '“', '★', '”', '–', '●', 'â', '►', '−', '¢', '²', '¬', '░', '¶', '↑', '±', '¿', '▾', '═', '¦', '║', '―', '¥', '▓', '—', '‹', '─', \n '▒', '：', '¼', '⊕', '▼', '▪', '†', '■', '’', '▀', '¨', '▄', '♫', '☆', 'é', '¯', '♦', '¤', '▲', 'è', '¸', '¾', 'Ã', '⋅', '‘', '∞', \n '∙', '）', '↓', '、', '│', '（', '»', '，', '♪', '╩', '╚', '³', '・', '╦', '╣', '╔', '╗', '▬', '❤', 'ï', 'Ø', '¹', '≤', '‡', '√', ]\ndef clean_text(x):\n    x = str(x)\n    for punct in puncts:\n        x = x.replace(punct, f' {punct} ')\n    return x","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true,"scrolled":false},"cell_type":"code","source":"train_df = pd.read_csv(\"../input/train.csv\")\ntest_df = pd.read_csv(\"../input/test.csv\")\nprint(\"Train shape : \",train_df.shape)\nprint(\"Test shape : \",test_df.shape)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"6ceb8ad2c73fd7d7e2e624d8d4abc89bee94cb54"},"cell_type":"code","source":"# Cleaning the data\ntrain_df[\"question_text\"] = train_df[\"question_text\"].apply(lambda x: clean_text(x))\ntest_df[\"question_text\"] = test_df[\"question_text\"].apply(lambda x: clean_text(x))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"7e712b37c05d1d88a00f605c688334adc8d7a1c8"},"cell_type":"code","source":"countries = [\"United States\", \"France\", \"India\", \"China\", \"Russia\", \"Saudi Arabia\", \"Iran\", \\\n             \"Germany\",\"England\", \"Europe\", \"Mexico\", \"Italy\", \"Spain\", \"Turkey\", \"Syria\", \"Israel\"]\n\npoliticians = [\"Trump\", \"Hillary\", \"Pence\", \"Obama\", \"Bush\", \"Clinton\", \"Putin\", \"Merkel\"]\n\ngroups = [\"Hispanics\", \"Blacks\", \"Whites\", \"Asians\", \"Browns\", \"Arabs\", \"Natives\", \"Caucasians\", \"Africans\",\\\n         \"Indians\", \"Chineses\", \"Mexicans\", \"British\", \"Americans\", \"Spanish\", \"Greek\", \"Italians\", \\\n          \"Turkish\", \"French\", \"English\", \"Germans\", \"Russians\",\"Arabic\", \"Persians\", \"European\", \"Madeup\"]\n\npolitical_groups = [\"democrats\", \"republicans\", \"liberals\", \"socialists\", \"communists\"]\n\nreligious_groups = [\"christians\", \"muslims\", \"jews\", \"buddhists\", \"hindus\", \"israelites\", \"zionists\"]\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f25645caca91c9aed7b811fdd6106e014deb2bed"},"cell_type":"code","source":"goofy_test_questions = []","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"6cfb7cc7569fedb24020e0172f8b27fd5f6f962e"},"cell_type":"code","source":"for country in countries:\n    goofy_test_questions.append(\"Why do not people see that \" + country + \" is a terrorist country?\")\n\nfor country1 in countries:\n    for country2 in countries:\n        goofy_test_questions.append(\"When will \" + country1 + \" end its occupation of \" + country2 + \"?\")\n        \nfor politician in politicians:\n    goofy_test_questions.append(\"How did an idiot like \" + politician + \" get elected?\")\n\nfor politician in politicians:\n    goofy_test_questions.append(\"How did a moron like \" + politician + \" get elected?\")\n\nfor politician in politicians:\n    goofy_test_questions.append(\"How did an incompetent officer like \" + politician + \" get elected?\")\n\nfor politician in politicians:\n    goofy_test_questions.append(\"How did \" + politician + \" get elected?\")\n\nfor politician in politicians:\n    goofy_test_questions.append(\"How does \" + politician + \" feel about the economic wreck he left behind?\")\n    \nfor group in groups:\n    goofy_test_questions.append(\"Are all \" + group + \" stupid?\")\n\nfor group in groups:\n    goofy_test_questions.append(\"Are \" + group + \" stupid?\")\n\nfor group in groups:\n    goofy_test_questions.append(\"Are all \" + group + \" rude?\")\n\nfor group in groups:\n    goofy_test_questions.append(\"Are \" + group + \" rude?\")\n\nfor country in countries:\n    goofy_test_questions.append(\"Why does \" + country + \" food taste so good?\")\n\nfor country in countries:\n    goofy_test_questions.append(\"Why does \" + country + \" food taste so bad?\")\n\nfor group in groups:\n    goofy_test_questions.append(\"Why do \" + group + \" people think themselves so high?\")\n\nfor group in groups:\n    goofy_test_questions.append(\"Should all \" + group + \" be spaded?\")\n\nfor group in groups:\n    goofy_test_questions.append(\"Should all \" + group + \" be sent to Mars and help build an infrastructure there to be useful for mankind and the world?\")\n\nfor group1 in groups:\n    for group2 in groups:\n        goofy_test_questions.append(\"What do \" + group1 + \" think of \" + group2 + \"?\")\n\nfor political_group in political_groups:\n    goofy_test_questions.append(\"Do \" + political_group + \" even care about the constitution?\")\n\nfor political_group in political_groups:\n    goofy_test_questions.append(\"Do \" + political_group + \" care about the country?\")\n\nfor religious_group in religious_groups:\n    goofy_test_questions.append(\"Do \" + religious_group + \" really believe in a book written by a child molester?\")\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"0f32b04b7a45e67fee3ba723492ad298923ace47"},"cell_type":"code","source":"goofy_test_y = np.ones(len(goofy_test_questions))\n\n\nfor sentence in goofy_test_questions:\n    sentence = clean_text(sentence)\n        \ngoofy_test_X = np.asarray(goofy_test_questions)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"41c4d7d386aaa7b50e977813a0b7b417f69a85ee"},"cell_type":"code","source":"## some config values \nembed_size = 300 # how big is each word vector\nmax_features = 90000 # how many unique words to use (i.e num rows in embedding vector)\nmaxlen = 60 # max number of words in a question to use","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"779bb89981facc0c885ebd09018d7c906eb4235b"},"cell_type":"code","source":"def get_long_sentences(sent_list, cut_off_len = maxlen):\n    ids_for_long = []\n    section_list = []\n    for i, sent in enumerate(list(sent_list)):\n        for section in re.split(r'([\\w+\\s?]+[.?!])', sent):\n            if len(section) > 1:\n                section_list.append(section)\n                ids_for_long.append(i)\n    return section_list, ids_for_long","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f121e3369a94889ea117b9e836f0294264efc6a4"},"cell_type":"code","source":"def get_generalized_sentences(sent_list, y_list):\n    global groups, political_groups, countries, religious_groups\n    general_X_list = []\n    general_y_list = []\n    for i, sent in enumerate(list(sent_list)):\n        for word in sent.split():\n            if (word in groups) or (word in political_groups) or (word in countries) or (word in religious_groups) :\n                new_sent = sent.replace(word, \"oovword\")\n                general_X_list.append(new_sent)\n                general_y_list.append(y_list[i])\n    return general_X_list, general_y_list","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"4488a5ecafd39c8f7c2253cb1ce815c080834ced"},"cell_type":"code","source":"def dot_product(x, kernel):\n    \"\"\"\n    Wrapper for dot product operation, in order to be compatible with both\n    Theano and Tensorflow\n    Args:\n        x (): input\n        kernel (): weights\n    Returns:\n    \"\"\"\n    if K.backend() == 'tensorflow':\n        return K.squeeze(K.dot(x, K.expand_dims(kernel)), axis=-1)\n    else:\n        return K.dot(x, kernel)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c485faa6f85066d5138e2121628c2f35bfacc1bd"},"cell_type":"code","source":"class AttentionWithContext(Layer):\n    \"\"\"\n    Attention operation, with a context/query vector, for temporal data.\n    Supports Masking.\n    Follows the work of Yang et al. [https://www.cs.cmu.edu/~diyiy/docs/naacl16.pdf]\n    \"Hierarchical Attention Networks for Document Classification\"\n    by using a context vector to assist the attention\n    # Input shape\n        3D tensor with shape: `(samples, steps, features)`.\n    # Output shape\n        2D tensor with shape: `(samples, features)`.\n    How to use:\n    Just put it on top of an RNN Layer (GRU/LSTM/SimpleRNN) with return_sequences=True.\n    The dimensions are inferred based on the output shape of the RNN.\n    Note: The layer has been tested with Keras 2.0.6\n    Example:\n        model.add(LSTM(64, return_sequences=True))\n        model.add(AttentionWithContext())\n        # next add a Dense layer (for classification/regression) or whatever...\n    \"\"\"\n\n    def __init__(self,\n                 W_regularizer=None, u_regularizer=None, b_regularizer=None,\n                 W_constraint=None, u_constraint=None, b_constraint=None,\n                 bias=True, **kwargs):\n\n\n        self.supports_masking = True\n        self.init = initializers.get('glorot_uniform')\n\n        self.W_regularizer = regularizers.get(W_regularizer)\n        self.u_regularizer = regularizers.get(u_regularizer)\n        self.b_regularizer = regularizers.get(b_regularizer)\n\n        self.W_constraint = constraints.get(W_constraint)\n        self.u_constraint = constraints.get(u_constraint)\n        self.b_constraint = constraints.get(b_constraint)\n\n        self.bias = bias\n        super(AttentionWithContext, self).__init__(**kwargs)\n\n\n    def build(self, input_shape):\n        assert len(input_shape) == 3\n\n        self.W = self.add_weight((input_shape[-1], input_shape[-1],),\n                                 initializer=self.init,\n                                 name='{}_W'.format(self.name),\n                                 regularizer=self.W_regularizer,\n                                 constraint=self.W_constraint)\n        if self.bias:\n            self.b = self.add_weight((input_shape[-1],),\n                                     initializer='zero',\n                                     name='{}_b'.format(self.name),\n                                     regularizer=self.b_regularizer,\n                                     constraint=self.b_constraint)\n\n        self.u = self.add_weight((input_shape[-1],),\n                                 initializer=self.init,\n                                 name='{}_u'.format(self.name),\n                                 regularizer=self.u_regularizer,\n                                 constraint=self.u_constraint)\n\n        super(AttentionWithContext, self).build(input_shape)\n\n    def compute_mask(self, input, input_mask=None):\n        # do not pass the mask to the next layers\n        return None\n\n    def call(self, x, mask=None):\n        uit = dot_product(x, self.W)\n\n        if self.bias:\n            uit += self.b\n\n        uit = K.tanh(uit)\n        ait = dot_product(uit, self.u)\n\n        a = K.exp(ait)\n\n        # apply mask after the exp. will be re-normalized next\n        if mask is not None:\n            # Cast the mask to floatX to avoid float64 upcasting in theano\n            a *= K.cast(mask, K.floatx())\n\n\n        # in some cases especially in the early stages of training the sum may be almost zero\n        # and this results in NaN's. A workaround is to add a very small positive number ε to the sum.\n        # a /= K.cast(K.sum(a, axis=1, keepdims=True), K.floatx())\n        a /= K.cast(K.sum(a, axis=1, keepdims=True) + K.epsilon(), K.floatx())\n\n        a = K.expand_dims(a)\n        weighted_input = x * a\n        return K.sum(weighted_input, axis=1)\n\n    def compute_output_shape(self, input_shape):\n        return input_shape[0], input_shape[-1]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"352ef15248c14690cb012256af6777f6e1bd007d"},"cell_type":"code","source":"## split to train and val\ntrain_df, val_df = train_test_split(train_df, test_size=0.1, random_state=2018)\n## Get the target values\ntrain_y = train_df['target'].values\nval_y = val_df['target'].values","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"9f62cda61722e18ee15c1b49bcd6febd7bce7536"},"cell_type":"code","source":"## fill up the missing values\ntrain_X = train_df[\"question_text\"].fillna(\"_na_\").values\nval_X = val_df[\"question_text\"].fillna(\"_na_\").values\ntest_X = test_df[\"question_text\"].fillna(\"_na_\").values","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c730cb5b4117912b321a8baf370d05ac288c0f5d"},"cell_type":"code","source":"generalized_train_X, generalized_train_y = get_generalized_sentences(train_X, train_y)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"65e1d0781f4305e96e367da16b470dd57323a374"},"cell_type":"code","source":"long_val_X, long_id_list = get_long_sentences(val_X)\nlong_test_X, long_test_id = get_long_sentences(test_X)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"fdafc0cf0b033355ae9820313e6bfa37dbb8554b"},"cell_type":"code","source":"## Tokenize the sentences\ntokenizer = Tokenizer(num_words=max_features)\n\ntokenizer_list = list(train_X)\n\ntokenizer.fit_on_texts(tokenizer_list)\ntrain_X = tokenizer.texts_to_sequences(train_X)\nval_X = tokenizer.texts_to_sequences(val_X)\ntest_X = tokenizer.texts_to_sequences(test_X)\n\nlong_val_X = tokenizer.texts_to_sequences(np.asarray(long_val_X))\nlong_test_X = tokenizer.texts_to_sequences(np.asarray(long_test_X))\n\ngeneralized_train_X = tokenizer.texts_to_sequences(np.asarray(generalized_train_X))\ngeneralized_train_y = np.asarray(generalized_train_y)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"617c3ffa91b81139b8923f6a205667213b1324c9"},"cell_type":"code","source":"goofy_test_X = tokenizer.texts_to_sequences(goofy_test_X)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"1770202abcf8d0cc7dfea99cb360eb25c7996c18"},"cell_type":"code","source":"## Pad the sentences for short\ntrain_X = pad_sequences(train_X, maxlen=maxlen)\nval_X = pad_sequences(val_X, maxlen=maxlen)\ntest_X = pad_sequences(test_X, maxlen=maxlen)\n\nlong_val_X = pad_sequences(long_val_X, maxlen=maxlen)\nlong_test_X = pad_sequences(long_test_X, maxlen=maxlen)\ngeneralized_train_X = pad_sequences(generalized_train_X, maxlen=maxlen)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b9216737d4ecd8f71b955bc29901e03eb21ccb1b"},"cell_type":"code","source":"goofy_test_X = pad_sequences(goofy_test_X, maxlen=maxlen)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"138703dd885423ae8b857cab7191ee8791c8c8e6"},"cell_type":"code","source":"EMBEDDING_FILE = '../input/embeddings/glove.840B.300d/glove.840B.300d.txt'\ndef get_coefs(word,*arr): return word, np.asarray(arr, dtype='float32')\nembeddings_index = dict(get_coefs(*o.split(\" \")) for o in open(EMBEDDING_FILE))\n\nall_embs = np.stack(embeddings_index.values())\nemb_mean,emb_std = all_embs.mean(), all_embs.std()\nembed_size = all_embs.shape[1]\n\nrandom_vector = np.random.rand(300)\n\nword_index = tokenizer.word_index\nnb_words = min(max_features, len(word_index))\nembedding_matrix1 = np.random.normal(emb_mean, emb_std, (nb_words, embed_size))\nfor word, i in word_index.items():\n    if i >= max_features: continue\n    embedding_vector = embeddings_index.get(word)\n    if embedding_vector is not None: \n        embedding_matrix1[i] = embedding_vector\n    else:\n        embedding_matrix1[i] = random_vector","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"aa4b95b3d801e5c631db99a2bea70a9decd4f57f"},"cell_type":"code","source":"EMBEDDING_FILE = '../input/embeddings/wiki-news-300d-1M/wiki-news-300d-1M.vec'\ndef get_coefs(word,*arr): return word, np.asarray(arr, dtype='float32')\nembeddings_index = dict(get_coefs(*o.split(\" \")) for o in open(EMBEDDING_FILE) if len(o)>100)\n\nall_embs = np.stack(embeddings_index.values())\nemb_mean,emb_std = all_embs.mean(), all_embs.std()\nembed_size = all_embs.shape[1]\n\nword_index = tokenizer.word_index\nnb_words = min(max_features, len(word_index))\nembedding_matrix2 = np.random.normal(emb_mean, emb_std, (nb_words, embed_size))\nfor word, i in word_index.items():\n    if i >= max_features: continue\n    embedding_vector = embeddings_index.get(word)\n    if embedding_vector is not None: embedding_matrix2[i] = embedding_vector","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d42363ff03b7c17a26a91604addc199203697722"},"cell_type":"code","source":"inp = Input(shape=(maxlen,))\nmodel1_out = Embedding(max_features, embed_size, weights=[embedding_matrix1],trainable=False)(inp)\nmodel1_out = Bidirectional(CuDNNGRU(128, return_sequences=True))(model1_out)\nmodel1_out = AttentionWithContext()(model1_out)\nmodel1_out = Dense(64, activation=\"relu\")(model1_out)\nmodel1_out = Dropout(0.1)(model1_out)\nmodel1_out = Dense(32, activation=\"relu\")(model1_out)\nmodel1_out = Dropout(0.1)(model1_out)\nmodel1_out = Dense(1, activation=\"sigmoid\")(model1_out)\nmodel = Model(inputs=inp, outputs=model1_out)\nmodel.compile(loss='binary_crossentropy', optimizer='adam', metrics=['accuracy'])\nprint(model.summary())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f2f242a71d5e7a48d2c8537dddae6524b752299a"},"cell_type":"code","source":"inp = Input(shape=(maxlen,))\nmodel2_out = Embedding(max_features, embed_size, weights=[embedding_matrix2],trainable=False)(inp)\nmodel2_out = Bidirectional(CuDNNGRU(128, return_sequences=True))(model2_out)\nmodel2_out = AttentionWithContext()(model2_out)\nmodel2_out = Dense(64, activation=\"relu\")(model2_out)\nmodel2_out = Dropout(0.1)(model2_out)\nmodel2_out = Dense(32, activation=\"relu\")(model2_out)\nmodel2_out = Dropout(0.1)(model2_out)\nmodel2_out = Dense(1, activation=\"sigmoid\")(model2_out)\nmodel2 = Model(inputs=inp, outputs=model2_out)\nmodel2.compile(loss='binary_crossentropy', optimizer='adam', metrics=['accuracy'])\nprint(model2.summary())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d75d100a03d7389e4a8cdf3495326785190fdcd7"},"cell_type":"code","source":"def train_model(model, all_train_X, all_train_y, all_val_X, all_val_y, epochs=2):\n    filepath=\"weights_best.h5\"\n    checkpoint = ModelCheckpoint(filepath, monitor='val_loss', verbose=2, save_best_only=True, mode='min')\n    reduce_lr = ReduceLROnPlateau(monitor='val_loss', factor=0.5, patience=1, min_lr=0.0001, verbose=2)\n    earlystopping = EarlyStopping(monitor='val_loss', min_delta=0.0001, patience=2, verbose=2, mode='auto')\n    callbacks = [checkpoint, reduce_lr]\n    for e in range(epochs):\n        model.fit(all_train_X, all_train_y, batch_size=1024, epochs=1, validation_data=(all_val_X, all_val_y), callbacks=callbacks)\n    model.load_weights(filepath)\n    return model","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c2699664faed980735d9029b3509881b5b956598"},"cell_type":"code","source":"model = train_model(model, train_X, train_y, val_X, val_y, epochs=12)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"2fa3425225503532db8babf3496bb635a49e98e4"},"cell_type":"code","source":"pred_val_y = model.predict(val_X, batch_size=1024, verbose=1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"1a048de2b9984250db108fe71c1f8d3f56a4df04"},"cell_type":"code","source":"model2 = train_model(model2, train_X, train_y, val_X, val_y, epochs=5)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"790a0ec6967eeb1755892b79c276e9047bf29f1b"},"cell_type":"code","source":"alt_pred_val_y = model2.predict(val_X, batch_size=1024, verbose=1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"bd903153175c7ddeed420c558fad71f58e5d7b81"},"cell_type":"code","source":"'''\nA function specific to this competition since the organizers don't want probabilities \nand only want 0/1 classification maximizing the F1 score. This function computes the best F1 score by looking at val set predictions\n'''\n\ndef f1_smart(y_true, y_pred):\n    thresholds = []\n    for thresh in np.arange(0.1, 0.901, 0.01):\n        thresh = np.round(thresh, 2)\n        res = metrics.f1_score(y_true, (y_pred > thresh).astype(int))\n        thresholds.append([thresh, res])\n        print(\"F1 score at threshold {0} is {1}\".format(thresh, res))\n\n    thresholds.sort(key=lambda x: x[1], reverse=True)\n    best_thresh = thresholds[0][0]\n    best_f1 = thresholds[0][1]\n    print(\"Best threshold: \", best_thresh)\n    return  best_f1, best_thresh","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d15afa4a2eec682d805b44ba8d0e4d6787418df0"},"cell_type":"code","source":"f1, threshold = f1_smart(val_y, pred_val_y)\nprint('Optimal F1: {} at threshold: {}'.format(f1, threshold))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"90e679e49d1977b6b71802c4f24fd916a5721add"},"cell_type":"code","source":"f1, alt_threshold = f1_smart(val_y, alt_pred_val_y)\nprint('Optimal F1: {} at threshold: {}'.format(f1, alt_threshold))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"78249f61b05f8cbcfd7e8140312f9b20991df21b"},"cell_type":"code","source":"long_pred_val_y = model.predict(long_val_X, batch_size=1024, verbose=1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"9fb7524f287a96cc6eea64bfea9e815ce2acb21b"},"cell_type":"code","source":"def update_preds(pred_val_y, long_pred_val_y, long_id_list):\n    global threshold\n    threshold_range = 0.10\n    copy_pred_val_y = pred_val_y\n    for i, pred in enumerate(list(pred_val_y)):\n        if (pred < (threshold + threshold_range)) and (pred > (threshold - threshold_range)):\n        # We do this extra step only if the model is not confident about the pred\n            count = 1\n            sum_pred = pred\n            for long_id, long_pred in zip(long_id_list, long_pred_val_y.tolist()):\n                if long_id == i:\n                    if (long_pred > (threshold + threshold_range)) or (long_pred < (threshold - threshold_range)):\n                        if long_pred[0] > pred:\n                    # We keep the pred that is closer to insincere                     \n                            copy_pred_val_y[i] = long_pred[0]\n    return copy_pred_val_y","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f315024366b210ba44c272b3513e8ff2b0164d5b"},"cell_type":"code","source":"pred_val_y = update_preds(pred_val_y, long_pred_val_y, long_id_list)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"4cb1568fa0c35bcc7027b7f2811d98bf70063235"},"cell_type":"code","source":"f1, threshold = f1_smart(val_y, pred_val_y)\nprint('Optimal F1: {} at threshold: {}'.format(f1, threshold))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a67ab1378e6bd0a39ff653830437b0518ef813c4"},"cell_type":"code","source":"all_train_X = np.concatenate((train_X,val_X),axis=0)\nall_train_y = np.concatenate((train_y,val_y),axis=0)\n\nall_train_X = np.concatenate((train_X,generalized_train_X),axis=0)\nall_train_y = np.concatenate((train_y,generalized_train_y),axis=0)\n\nmodel = train_model(model, train_X, train_y, val_X, val_y, epochs=5)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f09d768dc0db1805af31925e4fedce858fd6d6be"},"cell_type":"code","source":"def seq_ensemble(pred_y, alt_pred_y):\n    global threshold, alt_threshold\n    threshold_range_list = [0.15,0.12,0.10,0.08,0.05,0.03,0.02,0.01]\n    copy_pred_val_y = pred_val_y\n    for threshold_range in threshold_range_list:\n        for i, pred_pair in enumerate(zip(list(pred_val_y),list(alt_pred_y))):\n            if (pred_pair[0] < (threshold + threshold_range)) and \\\n                            (pred_pair[0] > (threshold - threshold_range)):\n                if (pred_pair[1] > (alt_threshold + threshold_range)) or \\\n                            (pred_pair[1] < (alt_threshold - threshold_range)):\n                    copy_pred_val_y[i] = pred_pair[1]    \n    return copy_pred_val_y","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"76f8a8fcc1ba8ec542bf758c22803eea942dd848"},"cell_type":"code","source":"pred_val_y = seq_ensemble(pred_val_y, alt_pred_val_y)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"572a5cccf1648eaa3759dcca6a04218fbdbd4f04"},"cell_type":"code","source":"f1, threshold = f1_smart(val_y, pred_val_y)\nprint('Optimal F1: {} at threshold: {}'.format(f1, threshold))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d08c5ab00b2e44aa7449ebf83ad8782e8d77a3e5"},"cell_type":"code","source":"pred_test_y = model.predict(test_X, batch_size=1024, verbose=1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"cc9c7bc40551417e8e261d4854ecf3f3a76c1e6f"},"cell_type":"code","source":"long_pred_test_y = model.predict(long_test_X, batch_size=1024, verbose=1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"7141ec1351bccba943da89d5790cba65d4deb422"},"cell_type":"code","source":"pred_test_y = update_preds(pred_test_y, long_pred_test_y, long_test_id)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d825cdc33fc6acf8d78b88dca160b7db340cebdd"},"cell_type":"code","source":"pred_test_y = (pred_test_y>threshold).astype(int)\nout_df = pd.DataFrame({\"qid\":test_df[\"qid\"].values})\nout_df['prediction'] = pred_test_y\nout_df.to_csv(\"submission.csv\", index=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"9c7e84f3d119b619a90ae2295b2066d7fe65221a"},"cell_type":"code","source":"pred_goofy_y = model.predict(goofy_test_X, batch_size=10, verbose=1)\nf1, threshold = f1_smart(goofy_test_y, pred_goofy_y)\nprint('Optimal F1: {} at threshold: {}'.format(f1, threshold))\n\nfor question,pred in zip(goofy_test_questions, pred_goofy_y): \n    print(pred, \" \", question, \"\\n\")","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}