{"cells":[{"metadata":{"_uuid":"edcc98e86754859e1f31a0b33f01b86148975cc1","trusted":true},"cell_type":"code","source":"import pandas as pd\nfrom tqdm import tqdm\nimport numpy as np\ntqdm.pandas()\n\nimport fastText as fasttext\nimport math\nfrom sklearn.model_selection import train_test_split\nfrom sklearn import metrics\nfrom sklearn.model_selection import GridSearchCV, StratifiedKFold\nfrom sklearn.metrics import f1_score, roc_auc_score\n\nfrom keras.preprocessing.text import Tokenizer\nfrom keras.preprocessing.sequence import pad_sequences\nfrom keras.layers import Dense, Input, CuDNNLSTM, Embedding, Dropout, Activation, CuDNNGRU, Conv1D\nfrom keras.layers import Bidirectional, GlobalMaxPool1D, GlobalMaxPooling1D, GlobalAveragePooling1D\nfrom keras.layers import Input, Embedding, Dense, Conv2D, MaxPool2D, concatenate\nfrom keras.layers import Reshape, Flatten, Concatenate, Dropout, SpatialDropout1D\nfrom keras.optimizers import Adam\nfrom keras.models import Model\nfrom keras import backend as K\nfrom keras.engine.topology import Layer\nfrom keras import initializers, regularizers, constraints, optimizers, layers\nfrom keras.layers import concatenate\nfrom keras.callbacks import *\nfrom keras.initializers import *\nfrom keras.optimizers import *\nfrom keras.callbacks import *\nfrom keras.layers import *\nfrom keras.models import *\nfrom nltk.tag import pos_tag\nfrom xgboost import XGBClassifier","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"1b1151984715f74225e3f987a9330b4fdd549835","trusted":true},"cell_type":"code","source":"## some config values \nembed_size = 300 # how big is each word vector\nmax_features = 95000 # how many unique words to use (i.e num rows in embedding vector)\nmaxlen = 70 # max number of words in a question to use\nTOP_N_FIRST_WORDS = 40 #originally 40\nNUM_CUSTOM_FEATURES = 15 + TOP_N_FIRST_WORDS\nNUM_TOTAL_FEATURES = 19 #after dropping nonimportant features","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"b10c1aaaba59d70c4df3ed3175b6fe84398f008e","trusted":true},"cell_type":"code","source":"#Loads embeddings\nglove_EMBEDDING_FILE = open('../input/embeddings/glove.840B.300d/glove.840B.300d.txt')\npara_embedding_file = open('../input/embeddings/paragram_300_sl999/paragram_300_sl999.txt', encoding=\"utf8\", errors='ignore')\n\ndef load_glove(word_index):\n    def get_coefs(word,*arr): return word, np.asarray(arr, dtype='float32')\n    embeddings_index = dict(get_coefs(*o.split(\" \")) for o in glove_EMBEDDING_FILE)\n\n    all_embs = np.stack(embeddings_index.values())\n    emb_mean,emb_std = all_embs.mean(), all_embs.std()\n    embed_size = all_embs.shape[1]\n\n    # word_index = tokenizer.word_index\n    nb_words = min(max_features, len(word_index))\n    embedding_matrix = np.random.normal(emb_mean, emb_std, (nb_words, embed_size))\n    #unique_words = [None] * max_features\n    for word, i in word_index.items():\n        if i >= max_features: continue\n        embedding_vector = embeddings_index.get(word)\n        #unique_words[i] = word\n        if embedding_vector is not None: embedding_matrix[i] = embedding_vector\n            \n    return embedding_matrix \n    \ndef load_fasttext(word_index):    \n    EMBEDDING_FILE = '../input/embeddings/wiki-news-300d-1M/wiki-news-300d-1M.vec'\n    def get_coefs(word,*arr): return word, np.asarray(arr, dtype='float32')\n    embeddings_index = dict(get_coefs(*o.split(\" \")) for o in open(EMBEDDING_FILE) if len(o)>100)\n\n    all_embs = np.stack(embeddings_index.values())\n    emb_mean,emb_std = all_embs.mean(), all_embs.std()\n    embed_size = all_embs.shape[1]\n\n    # word_index = tokenizer.word_index\n    nb_words = min(max_features, len(word_index))\n    embedding_matrix = np.random.normal(emb_mean, emb_std, (nb_words, embed_size))\n    #unique_words = [None] * max_features\n    for word, i in word_index.items():\n        if i >= max_features: continue\n        embedding_vector = embeddings_index.get(word)\n        #unique_words[i] = word\n        if embedding_vector is not None: embedding_matrix[i] = embedding_vector\n\n    return embedding_matrix\n\ndef load_para(word_index):\n    def get_coefs(word,*arr): return word, np.asarray(arr, dtype='float32')\n    embeddings_index = dict(get_coefs(*o.split(\" \")) for o in para_embedding_file if len(o)>100)\n\n    all_embs = np.stack(embeddings_index.values())\n    emb_mean,emb_std = all_embs.mean(), all_embs.std()\n    embed_size = all_embs.shape[1]\n\n    # word_index = tokenizer.word_index\n    nb_words = min(max_features, len(word_index))\n    embedding_matrix = np.random.normal(emb_mean, emb_std, (nb_words, embed_size))\n    #unique_words = [None] * max_features\n    for word, i in word_index.items():\n        if i >= max_features: continue\n        embedding_vector = embeddings_index.get(word)\n        #unique_words[i] = word\n        if embedding_vector is not None: embedding_matrix[i] = embedding_vector\n    \n    return embedding_matrix","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"afd570d1160b826a84c4c7af76950fd7a30e4471","trusted":true},"cell_type":"code","source":"#Text Cleaning\n\nimport re\n\ndef clean_numbers(x):\n\n    x = re.sub('[0-9]{5,}', '#####', x)\n    x = re.sub('[0-9]{4}', '####', x)\n    x = re.sub('[0-9]{3}', '###', x)\n    x = re.sub('[0-9]{2}', '##', x)\n    x = re.sub('[0-9]k', '# thousand', x)\n    return x","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"f1c3084cd7503c9148b681de897e0d38d7a90741","trusted":true},"cell_type":"code","source":"# Text Cleaning\ndef _get_mispell(mispell_dict):\n    mispell_re = re.compile('(%s)' % '|'.join(mispell_dict.keys()))\n    return mispell_dict, mispell_re\n\n\nmispell_dict = {'colour':'color',\n                'grey' : 'gray',\n                'recognised' : 'recognized',\n                'recognise' : 'recognize',\n                'defence' : 'defense',\n                'programmes' : 'programme',\n                'centre' : 'center',\n                'didnt' : 'did not',\n                'doesnt' : 'does not',\n                'isnt' : 'is not',\n                'Isnt' : 'is not',\n                'hasnt' : 'has not',\n                'wasnt' : 'was not',\n                'Doesnt' : 'does not',\n                'Shouldnt' : 'should not',\n                'shouldnt' : 'should not',\n                'favourite' : 'favorite',\n                'travelling' : 'traveling',\n                'counselling' : 'counseling',\n                'theatre' : 'theater',\n                'cancelled' : 'canceled',\n                'realized' : 'realised',\n                'memorise' : 'memorize',\n                'labour' : 'labor',\n                'organisation':'organization',\n                'wwii':'world war 2',\n                'citicise':'criticize',\n                'wwwyoutubecom' : 'youtube',\n                'instagram': 'social medium',\n                'whatsapp': 'social medium',\n                'WeChat' : 'social medium',\n                'snapchat': 'social medium',\n                'Snapchat' : 'social medium',\n                'Pinterest' : 'social medium',\n                'bitcoins' : 'cryptocurrency',\n                'bitcoin' : 'cryptocurrency',\n                'Ethereum' : 'cryptocurrency',\n                'cryptocurrencies' : 'cryptocurrency',\n                'ethereum' : 'cryptocurrency',\n                'Coinbase' : 'cryptocurrency',\n                'Blockchain' : 'cryptocurrency',\n                'Cryptocurrency' : 'cryptocurrency',\n                'Litecoin' : 'cryptocurrency',\n                'coinbase' : 'cryptocurrency',\n                'altcoin' : 'cryptocurrency',\n                'litecoin' : 'cryptocurrency',\n                'cryptos' : 'cryptocurrency',\n                'Fortnite' : 'game',\n                'Nodejs' : 'programming',\n                'nodejs' : 'programming',\n                'ReactJS' : 'programming',\n                'Golang' : 'programming',\n                'counsellor' : 'counselling',\n                'Tensorflow' : 'machine learning',\n                'TensorFlow' : 'machine learning',\n                'DeepMind' : 'machine learning',\n                'Codeforces' : 'programming',\n                'HackerRank' : 'programming',\n                'CodeChef' : 'programming',\n                'AngularJS' : 'programming',\n                'PewDiePie' : 'gaming social medium',\n                'brexit' : 'Brexit',\n                'Xiaomi' : 'phone company',\n                'OnePlus' : 'phone company',\n                'mastrubation' : 'masturbation',\n                'colour': 'color',\n                'centre': 'center',\n                'favourite': 'favorite',\n                'travelling': 'traveling',\n                'theatre': 'theater',\n                'cancelled': 'canceled',\n                'labour': 'labor',\n                'organisation': 'organization',\n                'wwii': 'world war 2',\n                'citicise': 'criticize',\n                'youtu ': 'youtube ',\n                'Qoura': 'Quora',\n                'sallary': 'salary',\n                'Whta': 'What',\n                'narcisist': 'narcissist',\n                'howdo': 'how do',\n                'whatare': 'what are',\n                'howcan': 'how can',\n                'howmuch': 'how much',\n                'howmany': 'how many',\n                'whydo': 'why do',\n                'doI': 'do I',\n                'theBest': 'the best',\n                'howdoes': 'how does',\n                'mastrubation': 'masturbation',\n                'mastrubate': 'masturbate',\n                \"mastrubating\": 'masturbating',\n                'pennis': 'penis',\n                'Etherium': 'cryptocurrency',\n                'narcissit': 'narcissist',\n                'bigdata': 'big data',\n                '2k17': '2017',\n                '2k18': '2018',\n                'qouta': 'quota',\n                'exboyfriend': 'ex boyfriend',\n                'airhostess': 'air hostess',\n                \"whst\": 'what',\n                'watsapp': 'whatsapp',\n                'demonitisation': 'demonetization',\n                'demonitization': 'demonetization',\n                'demonetisation': 'demonetization',\n                'mofo': 'fuck',\n                'ww2': 'world war 2',\n                'havent': 'have not',\n                'neonazis': 'Nazis',\n                'hillary clinton': 'Hillary Clinton',\n                'donald trump': 'Donald Trump'\n               }\n\n\n\nmispellings, mispellings_re = _get_mispell(mispell_dict)\n\ndef replace_typical_misspell(text):\n    def replace(match):\n        return mispellings[match.group(0)]\n\n    return mispellings_re.sub(replace, text)\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a90c58be82550cbdb75f64f84f29f39a56c31bfe"},"cell_type":"code","source":"import string\nstring.printable\nascii_chars = string.printable\nascii_chars += \" áéíóúàèìòùâêîôûäëïöüñõç\"\n\n#checks if a string of text contains any nonenglish characters (excluding punctuations, spanish, and french characters)\ndef contains_non_english(text):\n    if all(char in ascii_chars for char in text):\n        return 0\n    else:\n        return 1\n    \n#clean non english characters from string of text\ndef remove_non_english(text):\n    return ''.join(filter(lambda x: x in ascii_chars, text))\n\n\ndef get_first_word(word):\n    if(type(word) != \"float\"):\n        return word.split(\" \")[0]\n    return \"-1\"\n\ndef get_cap_vs_length(row):\n    if row[\"total_length\"] == 0:\n        return -1\n    return float(row['capitals'])/float(row['total_length'])\n\ndef calc_max_word_len(sentence):\n    maxLen = 0\n    for word in sentence:\n        maxLen = max(maxLen, len(word))\n    return maxLen\n\n#removes all single characters except for \"I\" and \"a\"\ndef remove_singles(text):\n    return ' '.join( [w for w in text.split() if ((len(w)>1) or (w.lower() == \"i\") or (w.lower() == \"a\"))] )\n    \n#combines multiple whitespaces into single\ndef clean_text(x):\n    x = str(x)\n    x = x.replace(\"-\", '')\n    x = x.replace(\"/\", '')\n    x = x.replace(\"'\",'')\n    x = re.sub( '\\s+', ' ', x).strip()\n    for punct in '&':\n        x = x.replace(punct, f' {punct} ')\n    for punct in '?!.,\"#$%\\'()*+-/:;<=>@[\\\\]^_`{|}~' + '“”’':\n        x = x.replace(punct, '')\n    return x","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"2e3152d920b1eb6f14b9d71159585d5832fb5c1a"},"cell_type":"code","source":"train = pd.read_csv(\"../input/train.csv\")\ntest = pd.read_csv(\"../input/test.csv\")\nprint(\"Train shape : \",train.shape)\nprint(\"Test shape : \",test.shape)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ae0fb9187aada7680dec501691cc8c0fe9f03b26"},"cell_type":"code","source":"# get number of bad words\ndef num_bad_words(text):\n    badwords = 0\n    input_words=text.split()\n    for word in input_words:\n        if word in bad_words:\n            badwords += 1\n    return badwords\n\n# get number of good words\ndef num_good_words(text):\n    goodwords = 0\n    input_words=text.split()\n    for word in input_words:\n        if word in good_words:\n            goodwords += 1\n    return goodwords","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"8f78ea7efb142e96dd19a94d255c9eac2ea0c733","trusted":true},"cell_type":"code","source":"#loads, generate features, then cleans\n\n#Generate features\n#df = pd.concat([train.loc[:, 'qid' : 'question_text'], test], sort = 'False')\n\nprint(\"--- Generating non_eng\")\ntrain[\"non_eng\"] = train[\"question_text\"].map(lambda x: contains_non_english(x))\ntest[\"non_eng\"] = test[\"question_text\"].map(lambda x: contains_non_english(x))\nprint(\"--- Generating first_word\")\ntrain[\"first_word\"] = train[\"question_text\"].map(lambda x: get_first_word(x))\ntest[\"first_word\"] = test[\"question_text\"].map(lambda x: get_first_word(x))\nprint(\"--- Generating total_length (num chars)\")\ntrain['total_length'] = train['question_text'].apply(len)\ntest['total_length'] = test['question_text'].apply(len)\nprint(\"--- Generating capitals\")\ntrain['capitals'] = train['question_text'].apply(lambda comment: sum(1 for c in comment if c.isupper()))\ntest['capitals'] = test['question_text'].apply(lambda comment: sum(1 for c in comment if c.isupper()))\n\nprint(\"--- Generating caps_vs_length\")\ntrain['caps_vs_length'] = train.apply(lambda row: get_cap_vs_length(row),axis=1)\ntest['caps_vs_length'] = test.apply(lambda row: get_cap_vs_length(row),axis=1)\n\n#print(\"--- Generating num_exclamation_marks\")\n#train['num_exclamation_marks'] = train['question_text'].apply(lambda comment: comment.count('!'))\n#test['num_exclamation_marks'] = test['question_text'].apply(lambda comment: comment.count('!'))\n\nprint(\"--- Generating num_question_marks\")\ntrain['num_question_marks'] = train['question_text'].apply(lambda comment: comment.count('?'))\ntest['num_question_marks'] = test['question_text'].apply(lambda comment: comment.count('?'))\n\nprint(\"--- Generating num_punctuation\")\ntrain['num_punctuation'] = train['question_text'].apply(lambda comment: sum(comment.count(w) for w in '.,;:'))\ntest['num_punctuation'] = test['question_text'].apply(lambda comment: sum(comment.count(w) for w in '.,;:'))\n\n#print(\"--- Generating num_symbols\")\n#train['num_symbols'] = train['question_text'].apply(lambda comment: sum(comment.count(w) for w in '*&$%'))\n#test['num_symbols'] = test['question_text'].apply(lambda comment: sum(comment.count(w) for w in '*&$%'))\n\nprint(\"--- Generating num_words\")\ntrain['num_words'] = train['question_text'].apply(lambda comment: len(re.sub(r'[^\\w\\s]','',comment).split(\" \")))\ntest['num_words'] = test['question_text'].apply(lambda comment: len(re.sub(r'[^\\w\\s]','',comment).split(\" \")))\n\nprint(\"--- Generating num_unique_words\")\ntrain['num_unique_words'] = train['question_text'].apply(lambda comment: len(set(w for w in comment.split())))\ntest['num_unique_words'] = test['question_text'].apply(lambda comment: len(set(w for w in comment.split())))\n\nprint(\"--- Generating words_vs_unique\")\ntrain['words_vs_unique'] = train['num_unique_words'] / train['num_words']\ntest['words_vs_unique'] = test['num_unique_words'] / test['num_words']\n\n#print(\"--- Generating num_smilies\")\n#train['num_smilies'] = train['question_text'].apply(lambda comment: sum(comment.count(w) for w in (':-)', ':)', ';-)', ';)')))\n#test['num_smilies'] = test['question_text'].apply(lambda comment: sum(comment.count(w) for w in (':-)', ':)', ';-)', ';)')))\n\nprint(\"--- Generating num_sentences\")\ntrain['num_sentences'] = train['question_text'].apply(lambda comment: len(re.split(r'[.!?]+', comment)))\ntest['num_sentences'] = test['question_text'].apply(lambda comment: len(re.split(r'[.!?]+', comment)))\n\nprint(\"--- Generating max_word_len\")\ntrain['max_word_len'] = train['question_text'].apply(lambda comment: calc_max_word_len(re.sub(r'[^\\w\\s]','',comment).split(\" \")))\ntest['max_word_len'] = test['question_text'].apply(lambda comment: calc_max_word_len(re.sub(r'[^\\w\\s]','',comment).split(\" \")))\n\nprint(\"cleaning text\")\ntrain[\"question_text\"] = train[\"question_text\"].apply(lambda x: clean_text(x))\ntest[\"question_text\"] = test[\"question_text\"].apply(lambda x: clean_text(x))\n\nprint(\"remove single characters\")\ntrain[\"question_text\"] = train[\"question_text\"].apply(lambda x: remove_singles(x))\ntest[\"question_text\"] = test[\"question_text\"].apply(lambda x: remove_singles(x))\n\nprint(\"cleaning numbers\")\ntrain[\"question_text\"] = train[\"question_text\"].apply(lambda x: clean_numbers(x))\ntest[\"question_text\"] = test[\"question_text\"].apply(lambda x: clean_numbers(x))\n\nprint(\"cleaning misspellings\")\ntrain[\"question_text\"] = train[\"question_text\"].apply(lambda x: replace_typical_misspell(x))\ntest[\"question_text\"] = test[\"question_text\"].apply(lambda x: replace_typical_misspell(x))\n\nprint(\"filling missing values\")\n#clean chinese, korean, japanese characters\nprint('cleaning characters')\ntrain[\"question_text\"] = train[\"question_text\"].map(lambda x: remove_non_english(x))\ntest[\"question_text\"] = test[\"question_text\"].map(lambda x: remove_non_english(x))\n\n## fill up the missing values\ntrain[\"question_text\"].fillna(\"\").values\ntest[\"question_text\"].fillna(\"\").values\n\n#for getting num good and bad words\nfrom wordcloud import STOPWORDS\nfrom collections import defaultdict\nimport operator\n\ntrain1_df = train[train[\"target\"]==1]\ntrain0_df = train[train[\"target\"]==0]\n\n## custom function for ngram generation ##\ndef generate_ngrams(text, n_gram=1):\n    token = [token for token in text.lower().split(\" \") if token != \"\" if token not in STOPWORDS]\n    ngrams = zip(*[token[i:] for i in range(n_gram)])\n    return [\" \".join(ngram) for ngram in ngrams]\n\nfreq_dict_bad = defaultdict(int)\nfor sent in train1_df[\"question_text\"]:\n    for word in generate_ngrams(sent):\n        freq_dict_bad[word] += 1\nfreq_dict_bad = dict(freq_dict_bad)\n\nfreq_dict_good = defaultdict(int)\nfor sent in train0_df[\"question_text\"]:\n    for word in generate_ngrams(sent):\n        freq_dict_good[word] += 1\nfreq_dict_good = dict(freq_dict_good)\n\nbad_words = sorted(freq_dict_bad, key=freq_dict_bad.get, reverse=True)[:1000]\ngood_words = sorted(freq_dict_good, key=freq_dict_good.get, reverse=True)[:1000]\n\nprint(\"--- Generating num_bad_words\")\ntrain[\"num_bad_words\"] = train[\"question_text\"].map(lambda x: num_bad_words(x))\ntest[\"num_bad_words\"] = test[\"question_text\"].map(lambda x: num_bad_words(x))\n\nprint(\"--- Generating num_good_words\")\ntrain[\"num_good_words\"] = train[\"question_text\"].map(lambda x: num_good_words(x))\ntest[\"num_good_words\"] = test[\"question_text\"].map(lambda x: num_good_words(x))\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"576fa8cd1c7d3b2416227baf371dadc91a882e6b"},"cell_type":"code","source":"top_x = ['What', 'How', 'Why', 'Is', 'Can', 'Which', 'Do', 'If', 'Are', 'Who', 'I', 'Does', 'Where', 'Should', 'When', 'Will', \"What's\", 'In', 'Would', 'Have', 'Did', 'My', 'Has', 'As', 'Could', \"I'm\", 'Was', 'A', 'The', 'What’s', 'For', 'After', 'At', 'With', 'Am', 'Were', 'From', 'Since', 'To', 'On']\ntrain[\"first_word\"] = train[\"first_word\"].map(lambda x: x if x in top_x else \"Other\")\ntest[\"first_word\"] = test[\"first_word\"].map(lambda x: x if x in top_x else \"Other\")\none_hot_encoded_first_word_train = pd.get_dummies(train[\"first_word\"])\none_hot_encoded_first_word_test = pd.get_dummies(test[\"first_word\"])\n\noriginal_headers_train = list(train.columns.values)\noriginal_headers_test = list(test.columns.values)\none_hot_encoded_first_word_headers_train = one_hot_encoded_first_word_train.columns.values\none_hot_encoded_first_word_headers_test = one_hot_encoded_first_word_test.columns.values\none_hot_encoded_first_word_headers_train = [\"first_word_\" + x for x in one_hot_encoded_first_word_headers_train]\none_hot_encoded_first_word_headers_test = [\"first_word_\" + x for x in one_hot_encoded_first_word_headers_test]\nnew_train_headers = original_headers_train + one_hot_encoded_first_word_headers_train\nnew_test_headers = original_headers_test + one_hot_encoded_first_word_headers_test\n\ntrain_with_features = pd.concat([train, one_hot_encoded_first_word_test], axis=1, ignore_index=True)\ntrain_with_features.columns = new_train_headers\n\ntest_with_features =  pd.concat([test, one_hot_encoded_first_word_test], axis=1, ignore_index=True)\ntest_with_features.columns = new_test_headers","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b2f7065f920dc7331de0590df6cb325c109c04dc"},"cell_type":"code","source":"#drop first word features that are insignificant\nto_drop = [\n       'first_word_Other', 'first_word_Can', 'first_word_Where',\n       'first_word_Since', 'first_word_Did',\n       'first_word_If', \"first_word_What's\", 'first_word_Should',\n       'first_word_Who', 'first_word_I', 'first_word_Was',\n       'first_word_The', 'first_word_When', 'first_word_Is',\n       'first_word_Would', 'first_word_In', 'first_word_As',\n       'first_word_Were', 'first_word_Will', 'first_word_What’s',\n       'first_word_My', 'first_word_With', 'first_word_Am',\n       'first_word_To', 'first_word_At', 'first_word_From',\n       'first_word_Has', 'first_word_After',\n       'first_word_Does', 'first_word_Could', 'first_word_A',\n       \"first_word_I'm\", 'first_word_For', 'first_word_Have',\n       'first_word_On']\n\ntrain_with_features = train_with_features.drop(to_drop,axis=1)\ntest_with_features = test_with_features.drop(to_drop,axis=1)\n","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"c09a8e14529a448161e25ed5f87ff18cbd1af1ef","trusted":true},"cell_type":"code","source":"#Tokenizes the data\ndef tokenize():\n\n    ## fill up the missing values\n    train_X = train[\"question_text\"]\n    test_X = test[\"question_text\"]\n\n    ## Tokenize the sentences\n    tokenizer = Tokenizer(num_words=max_features)\n    tokenizer.fit_on_texts(list(train_X))\n    train_X = tokenizer.texts_to_sequences(train_X)\n    test_X = tokenizer.texts_to_sequences(test_X)\n\n    ## Pad the sentences \n    train_X = pad_sequences(train_X, maxlen=maxlen)\n    test_X = pad_sequences(test_X, maxlen=maxlen)\n\n    ## Get the target values\n    train_y = train['target'].values\n    \n    #shuffling the data\n    np.random.seed(2018)\n    trn_idx = np.random.permutation(len(train_X))\n\n    train_X = train_X[trn_idx]\n    train_y = train_y[trn_idx]\n    \n    return train_X, test_X, train_y, tokenizer.word_index\n\n'''def replace_words(sentence):\n    new_sentence = \"\"\n    for x in sentence.split(\" \"):\n        if(x in bad_words):\n            new_sentence += \"bacon \"\n        else if(x in good_words):\n            new_sentence += \"eggs \"\n        else if(x in both_words):\n            new_sentence += \"ham \"\n        else:\n            new_sentence += x + \" \"\n\ndef simple_tokenize():\n    ## fill up the missing values\n    train_X = train[\"question_text\"]\n    test_X = test[\"question_text\"]\n    \n    print(\"removing bad / good words for train\")\n    train_X[\"question_text\"] = train_X[\"question_text\"].apply(lambda x: replace_words(x))\n    print(\"removing bad / good words for test\")\n    test_X[\"question_text\"] = test_X[\"question_text\"].apply(lambda x: replace_words(x))\n\n    ## Tokenize the sentences\n    tokenizer = Tokenizer(num_words=max_features)\n    tokenizer.fit_on_texts(list(train_X))\n    train_X = tokenizer.texts_to_sequences(train_X)\n    print(train_X)\n    test_X = tokenizer.texts_to_sequences(test_X)\n\n    ## Pad the sentences \n    train_X = pad_sequences(train_X, maxlen=maxlen)\n    test_X = pad_sequences(test_X, maxlen=maxlen)\n\n    ## Get the target values\n    train_y = train['target'].values\n    \n    #shuffling the data\n    np.random.seed(2018)\n    trn_idx = np.random.permutation(len(train_X))\n\n    train_X = train_X[trn_idx]\n    train_y = train_y[trn_idx]\n    \n    return train_X, test_X, train_y, tokenizer.word_index '''","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"ec2ab81b72111600948eb3e4bab937e5f7f76476","trusted":true},"cell_type":"code","source":"# Create our own embedding that take average of 3 embeddings as this is better than concatenating embeddings\ntrain_X, test_X, train_y, word_index = tokenize()\nembedding_matrix_1 = load_glove(word_index)\n#unique_words2, embedding_matrix_2 = load_fasttext(word_index)\nembedding_matrix_3 = load_para(word_index)\n\nembedding_matrix = np.mean([embedding_matrix_1, embedding_matrix_3], axis = 0)\nnp.shape(embedding_matrix)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"45c5929fd11ea1a86ed9112b38be0e96b5397a34","trusted":true,"scrolled":true},"cell_type":"code","source":"\ndf = pd.concat([train_with_features.drop(['target'], axis=1), test], sort = 'False')\nprint(df.shape)\n#TODO: REMOVE THIS LINE (tests on only first 300)\n#df = df[:300]\n#train = train[:300]\n#train_X = train_X[:300]\n#train_y = train_y[:300]\n","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"ad05f9587adc19c3d9dd5e5b1b13880b15d48e4a","trusted":true},"cell_type":"code","source":"# len([ \"total_length\", \"capitals\",\"caps_vs_length\", \"num_exclamation_marks\",\"num_question_marks\",\"num_punctuation\",\"num_symbols\",\"num_words\",\"num_unique_words\",\"words_vs_unique\",\"num_smilies\"])\n#Normalize feature values\nfeatures_names = df.drop([\"qid\",\"question_text\",\"first_word\"],axis=1).columns.values\ntmp_df = df[features_names]\n\nfrom sklearn.preprocessing import MinMaxScaler\nscaler = MinMaxScaler()\ntmp_df = scaler.fit_transform(tmp_df)\n\ntrain_data = tmp_df[0 : train.shape[0]]\ntest_data  = tmp_df[train.shape[0] : (train.shape[0] + test.shape[0])]\n\n\nx_features_train = train_data\nx_features_test  = test_data\npd.DataFrame(x_features_train).head()\n#train_data.shape","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"db5739b4887f47ef03991dadf10d632bcb8f0c86","trusted":true},"cell_type":"code","source":"#The Model itself\n'''def model_lstm_atten(embedding_matrix):\n    \n    input_embeddings = Input(shape=(maxlen,))\n    x = Embedding(max_features, embed_size, weights=[embedding_matrix], trainable=False)(input_embeddings)\n    x = SpatialDropout1D(0.1)(x)\n    x = Bidirectional(CuDNNLSTM(40, return_sequences=True))(x)\n    y = Bidirectional(CuDNNGRU(40, return_sequences=True))(x)\n    \n    atten_1 = Attention(maxlen)(x) # skip connect\n    atten_2 = Attention(maxlen)(y)\n    avg_pool = GlobalAveragePooling1D()(y)\n    max_pool = GlobalMaxPooling1D()(y)\n    \n    input_features = Input(shape = (NUM_CUSTOM_FEATURES,))\n    conc = concatenate([atten_1, atten_2, avg_pool, max_pool, input_features])\n    conc = Dense(16, activation=\"relu\")(conc)\n    conc = Dropout(0.1)(conc)\n    outp = Dense(1, activation=\"sigmoid\")(conc)    \n\n    model = Model(inputs=[input_embeddings, input_features], outputs=outp)\n    model.compile(loss='binary_crossentropy', optimizer='adam', metrics=[f1])\n    \n    return model'''\n#The Model itself\ndef model_gru_cap(embedding_matrix):\n    \n    input_embeddings = Input(shape=(maxlen,))\n    x = Embedding(max_features, embed_size, weights=[embedding_matrix], trainable=False)(input_embeddings)\n    x = SpatialDropout1D(0.2)(x)\n    x = Bidirectional(CuDNNGRU(100, return_sequences=True, \n                                kernel_initializer=glorot_normal(seed=12300), recurrent_initializer=orthogonal(gain=1.0, seed=10000)))(x)\n    \n    x = Capsule(num_capsule=10, dim_capsule=10, routings=4, share_weights=True)(x)\n    x = Flatten()(x)\n    \n    '''atten_1 = Attention(maxlen)(x) # skip connect\n    atten_2 = Attention(maxlen)(y)\n    avg_pool = GlobalAveragePooling1D()(y)\n    max_pool = GlobalMaxPooling1D()(y)'''\n    \n    input_features = Input(shape = (NUM_TOTAL_FEATURES,))\n    #conc = concatenate([atten_1, atten_2, avg_pool, max_pool, input_features])\n    #conc = Dense(16, activation=\"relu\")(conc)\n    #conc = Dropout(0.1)(conc)\n    x = Dense(100, activation=\"relu\", kernel_initializer=glorot_normal(seed=12300))(x)\n    x = Dropout(0.12)(x)\n    x = BatchNormalization()(x)\n    \n    outp = Dense(1, activation=\"sigmoid\")(x)    \n\n    model = Model(inputs=[input_embeddings, input_features], outputs=outp)\n    model.compile(loss='binary_crossentropy', optimizer='adam', metrics=[f1])\n    \n    return model","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"9abd048a9a93fa9a520a030b491bc03d58e494b4","trusted":true},"cell_type":"code","source":"# https://www.kaggle.com/ryanzhang/tfidf-naivebayes-logreg-baseline\n\ndef threshold_search(y_true, y_proba):\n    best_threshold = 0\n    best_score = 0\n    for threshold in [i * 0.01 for i in range(100)]:\n        score = f1_score(y_true=y_true, y_pred=y_proba > threshold)\n        if score > best_score:\n            best_threshold = threshold\n            best_score = score\n    search_result = {'threshold': best_threshold, 'f1': best_score}\n    return search_result","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"f36ff1ffbb2319175a76d02608aa16f6ecb7d29f","trusted":true},"cell_type":"code","source":"# https://www.kaggle.com/suicaokhoailang/lstm-attention-baseline-0-652-lb\n#Attention and CyclicLR and Capsule\n\ndef squash(x, axis=-1):\n    # s_squared_norm is really small\n    # s_squared_norm = K.sum(K.square(x), axis, keepdims=True) + K.epsilon()\n    # scale = K.sqrt(s_squared_norm)/ (0.5 + s_squared_norm)\n    # return scale * x\n    s_squared_norm = K.sum(K.square(x), axis, keepdims=True)\n    scale = K.sqrt(s_squared_norm + K.epsilon())\n    return x / scale\n\n# A Capsule Implement with Pure Keras\nclass Capsule(Layer):\n    def __init__(self, num_capsule, dim_capsule, routings=3, kernel_size=(9, 1), share_weights=True,\n                 activation='default', **kwargs):\n        super(Capsule, self).__init__(**kwargs)\n        self.num_capsule = num_capsule\n        self.dim_capsule = dim_capsule\n        self.routings = routings\n        self.kernel_size = kernel_size\n        self.share_weights = share_weights\n        if activation == 'default':\n            self.activation = squash\n        else:\n            self.activation = Activation(activation)\n\n    def build(self, input_shape):\n        super(Capsule, self).build(input_shape)\n        input_dim_capsule = input_shape[-1]\n        if self.share_weights:\n            self.W = self.add_weight(name='capsule_kernel',\n                                     shape=(1, input_dim_capsule,\n                                            self.num_capsule * self.dim_capsule),\n                                     # shape=self.kernel_size,\n                                     initializer='glorot_uniform',\n                                     trainable=True)\n        else:\n            input_num_capsule = input_shape[-2]\n            self.W = self.add_weight(name='capsule_kernel',\n                                     shape=(input_num_capsule,\n                                            input_dim_capsule,\n                                            self.num_capsule * self.dim_capsule),\n                                     initializer='glorot_uniform',\n                                     trainable=True)\n\n    def call(self, u_vecs):\n        if self.share_weights:\n            u_hat_vecs = K.conv1d(u_vecs, self.W)\n        else:\n            u_hat_vecs = K.local_conv1d(u_vecs, self.W, [1], [1])\n\n        batch_size = K.shape(u_vecs)[0]\n        input_num_capsule = K.shape(u_vecs)[1]\n        u_hat_vecs = K.reshape(u_hat_vecs, (batch_size, input_num_capsule,\n                                            self.num_capsule, self.dim_capsule))\n        u_hat_vecs = K.permute_dimensions(u_hat_vecs, (0, 2, 1, 3))\n        # final u_hat_vecs.shape = [None, num_capsule, input_num_capsule, dim_capsule]\n\n        b = K.zeros_like(u_hat_vecs[:, :, :, 0])  # shape = [None, num_capsule, input_num_capsule]\n        for i in range(self.routings):\n            b = K.permute_dimensions(b, (0, 2, 1))  # shape = [None, input_num_capsule, num_capsule]\n            c = K.softmax(b)\n            c = K.permute_dimensions(c, (0, 2, 1))\n            b = K.permute_dimensions(b, (0, 2, 1))\n            outputs = self.activation(tf.keras.backend.batch_dot(c, u_hat_vecs, [2, 2]))\n            if i < self.routings - 1:\n                b = tf.keras.backend.batch_dot(outputs, u_hat_vecs, [2, 3])\n\n        return outputs\n\n    def compute_output_shape(self, input_shape):\n        return (None, self.num_capsule, self.dim_capsule)\n\nclass Attention(Layer):\n    def __init__(self, step_dim,\n                 W_regularizer=None, b_regularizer=None,\n                 W_constraint=None, b_constraint=None,\n                 bias=True, **kwargs):\n        self.supports_masking = True\n        self.init = initializers.get('glorot_uniform')\n\n        self.W_regularizer = regularizers.get(W_regularizer)\n        self.b_regularizer = regularizers.get(b_regularizer)\n\n        self.W_constraint = constraints.get(W_constraint)\n        self.b_constraint = constraints.get(b_constraint)\n\n        self.bias = bias\n        self.step_dim = step_dim\n        self.features_dim = 0\n        super(Attention, self).__init__(**kwargs)\n\n    def build(self, input_shape):\n        assert len(input_shape) == 3\n\n        self.W = self.add_weight((input_shape[-1],),\n                                 initializer=self.init,\n                                 name='{}_W'.format(self.name),\n                                 regularizer=self.W_regularizer,\n                                 constraint=self.W_constraint)\n        self.features_dim = input_shape[-1]\n\n        if self.bias:\n            self.b = self.add_weight((input_shape[1],),\n                                     initializer='zero',\n                                     name='{}_b'.format(self.name),\n                                     regularizer=self.b_regularizer,\n                                     constraint=self.b_constraint)\n        else:\n            self.b = None\n\n        self.built = True\n\n    def compute_mask(self, input, input_mask=None):\n        return None\n\n    def call(self, x, mask=None):\n        features_dim = self.features_dim\n        step_dim = self.step_dim\n\n        eij = K.reshape(K.dot(K.reshape(x, (-1, features_dim)),\n                        K.reshape(self.W, (features_dim, 1))), (-1, step_dim))\n\n        if self.bias:\n            eij += self.b\n\n        eij = K.tanh(eij)\n\n        a = K.exp(eij)\n\n        if mask is not None:\n            a *= K.cast(mask, K.floatx())\n\n        a /= K.cast(K.sum(a, axis=1, keepdims=True) + K.epsilon(), K.floatx())\n\n        a = K.expand_dims(a)\n        weighted_input = x * a\n        return K.sum(weighted_input, axis=1)\n\n    def compute_output_shape(self, input_shape):\n        return input_shape[0],  self.features_dim\n\n# https://www.kaggle.com/hireme/fun-api-keras-f1-metric-cyclical-learning-rate/code\n\nclass CyclicLR(Callback):\n    \"\"\"This callback implements a cyclical learning rate policy (CLR).\n    The method cycles the learning rate between two boundaries with\n    some constant frequency, as detailed in this paper (https://arxiv.org/abs/1506.01186).\n    The amplitude of the cycle can be scaled on a per-iteration or \n    per-cycle basis.\n    This class has three built-in policies, as put forth in the paper.\n    \"triangular\":\n        A basic triangular cycle w/ no amplitude scaling.\n    \"triangular2\":\n        A basic triangular cycle that scales initial amplitude by half each cycle.\n    \"exp_range\":\n        A cycle that scales initial amplitude by gamma**(cycle iterations) at each \n        cycle iteration.\n    For more detail, please see paper.\n    \n    # Example\n        ```python\n            clr = CyclicLR(base_lr=0.001, max_lr=0.006,\n                                step_size=2000., mode='triangular')\n            model.fit(X_train, Y_train, callbacks=[clr])\n        ```\n    \n    Class also supports custom scaling functions:\n        ```python\n            clr_fn = lambda x: 0.5*(1+np.sin(x*np.pi/2.))\n            clr = CyclicLR(base_lr=0.001, max_lr=0.006,\n                                step_size=2000., scale_fn=clr_fn,\n                                scale_mode='cycle')\n            model.fit(X_train, Y_train, callbacks=[clr])\n        ```    \n    # Arguments\n        base_lr: initial learning rate which is the\n            lower boundary in the cycle.\n        max_lr: upper boundary in the cycle. Functionally,\n            it defines the cycle amplitude (max_lr - base_lr).\n            The lr at any cycle is the sum of base_lr\n            and some scaling of the amplitude; therefore \n            max_lr may not actually be reached depending on\n            scaling function.\n        step_size: number of training iterations per\n            half cycle. Authors suggest setting step_size\n            2-8 x training iterations in epoch.\n        mode: one of {triangular, triangular2, exp_range}.\n            Default 'triangular'.\n            Values correspond to policies detailed above.\n            If scale_fn is not None, this argument is ignored.\n        gamma: constant in 'exp_range' scaling function:\n            gamma**(cycle iterations)\n        scale_fn: Custom scaling policy defined by a single\n            argument lambda function, where \n            0 <= scale_fn(x) <= 1 for all x >= 0.\n            mode paramater is ignored \n        scale_mode: {'cycle', 'iterations'}.\n            Defines whether scale_fn is evaluated on \n            cycle number or cycle iterations (training\n            iterations since start of cycle). Default is 'cycle'.\n    \"\"\"\n\n    def __init__(self, base_lr=0.001, max_lr=0.006, step_size=2000., mode='triangular',\n                 gamma=1., scale_fn=None, scale_mode='cycle'):\n        super(CyclicLR, self).__init__()\n\n        self.base_lr = base_lr\n        self.max_lr = max_lr\n        self.step_size = step_size\n        self.mode = mode\n        self.gamma = gamma\n        if scale_fn == None:\n            if self.mode == 'triangular':\n                self.scale_fn = lambda x: 1.\n                self.scale_mode = 'cycle'\n            elif self.mode == 'triangular2':\n                self.scale_fn = lambda x: 1/(2.**(x-1))\n                self.scale_mode = 'cycle'\n            elif self.mode == 'exp_range':\n                self.scale_fn = lambda x: gamma**(x)\n                self.scale_mode = 'iterations'\n        else:\n            self.scale_fn = scale_fn\n            self.scale_mode = scale_mode\n        self.clr_iterations = 0.\n        self.trn_iterations = 0.\n        self.history = {}\n\n        self._reset()\n\n    def _reset(self, new_base_lr=None, new_max_lr=None,\n               new_step_size=None):\n        \"\"\"Resets cycle iterations.\n        Optional boundary/step size adjustment.\n        \"\"\"\n        if new_base_lr != None:\n            self.base_lr = new_base_lr\n        if new_max_lr != None:\n            self.max_lr = new_max_lr\n        if new_step_size != None:\n            self.step_size = new_step_size\n        self.clr_iterations = 0.\n        \n    def clr(self):\n        cycle = np.floor(1+self.clr_iterations/(2*self.step_size))\n        x = np.abs(self.clr_iterations/self.step_size - 2*cycle + 1)\n        if self.scale_mode == 'cycle':\n            return self.base_lr + (self.max_lr-self.base_lr)*np.maximum(0, (1-x))*self.scale_fn(cycle)\n        else:\n            return self.base_lr + (self.max_lr-self.base_lr)*np.maximum(0, (1-x))*self.scale_fn(self.clr_iterations)\n        \n    def on_train_begin(self, logs={}):\n        logs = logs or {}\n\n        if self.clr_iterations == 0:\n            K.set_value(self.model.optimizer.lr, self.base_lr)\n        else:\n            K.set_value(self.model.optimizer.lr, self.clr())        \n            \n    def on_batch_end(self, epoch, logs=None):\n        \n        logs = logs or {}\n        self.trn_iterations += 1\n        self.clr_iterations += 1\n\n        self.history.setdefault('lr', []).append(K.get_value(self.model.optimizer.lr))\n        self.history.setdefault('iterations', []).append(self.trn_iterations)\n\n        for k, v in logs.items():\n            self.history.setdefault(k, []).append(v)\n        \n        K.set_value(self.model.optimizer.lr, self.clr())\n    \n\ndef f1(y_true, y_pred):\n    '''\n    metric from here \n    https://stackoverflow.com/questions/43547402/how-to-calculate-f1-macro-in-keras\n    '''\n    def recall(y_true, y_pred):\n        \"\"\"Recall metric.\n\n        Only computes a batch-wise average of recall.\n\n        Computes the recall, a metric for multi-label classification of\n        how many relevant items are selected.\n        \"\"\"\n        true_positives = K.sum(K.round(K.clip(y_true * y_pred, 0, 1)))\n        possible_positives = K.sum(K.round(K.clip(y_true, 0, 1)))\n        recall = true_positives / (possible_positives + K.epsilon())\n        return recall\n\n    def precision(y_true, y_pred):\n        \"\"\"Precision metric.\n\n        Only computes a batch-wise average of precision.\n\n        Computes the precision, a metric for multi-label classification of\n        how many selected items are relevant.\n        \"\"\"\n        true_positives = K.sum(K.round(K.clip(y_true * y_pred, 0, 1)))\n        predicted_positives = K.sum(K.round(K.clip(y_pred, 0, 1)))\n        precision = true_positives / (predicted_positives + K.epsilon())\n        return precision\n    precision = precision(y_true, y_pred)\n    recall = recall(y_true, y_pred)\n    return 2*((precision*recall)/(precision+recall+K.epsilon()))\n","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"2c5534e5ccd3d64737ef255a7061ade3ebcb184c","trusted":true},"cell_type":"code","source":"DATA_SPLIT_SEED = 2018\nNUM_SPLITS = 4\nclr = CyclicLR(base_lr=0.001, max_lr=0.002,\n               step_size=300., mode='exp_range',\n               gamma=0.99994)\n\ntrain_meta = np.zeros(train_y.shape)\ntest_meta = np.zeros(test_X.shape[0])\nsplits = list(StratifiedKFold(n_splits=NUM_SPLITS, shuffle=True, random_state=DATA_SPLIT_SEED).split(train_X, train_y))\nfeatures_split = list(StratifiedKFold(n_splits=NUM_SPLITS, shuffle=True, random_state=DATA_SPLIT_SEED).split(x_features_train, train_y))","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"fcad97019380b5d9a537c816134e375c47eef190","trusted":true},"cell_type":"code","source":"def train_pred(model, train_X, train_y, val_X, val_y, train_features, val_features, epochs=6, callback=None):\n    for e in range(epochs):\n        model.fit(x = [train_X, train_features], y = train_y, batch_size=512, epochs=1, validation_data=([val_X, val_features], val_y), callbacks = callback, verbose=0)\n        pred_val_y = model.predict([val_X, val_features], batch_size=1024, verbose=0)\n        \n        \n        '''print(pred_val_y)\n        print(len(pred_val_y))\n        print(val_y)\n        print(len(val_y))\n        print(set(val_y.flatten()) - set(pred_val_y.flatten()))\n        print((pred_val_y > 0.33).astype(int))'''\n        \n        best_score = metrics.f1_score(val_y, (pred_val_y > 0.33).astype(int))\n        print(\"Epoch: \", e, \"-    Val F1 Score: {:.4f}\".format(best_score))\n\n    pred_test_y = model.predict([test_X, x_features_test], batch_size=1024, verbose=0)\n    print('=' * 60)\n    return pred_val_y, pred_test_y, best_score","execution_count":null,"outputs":[]},{"metadata":{"_kg_hide-output":false,"_uuid":"6c5cf4088877c0aed21afd22b7cc4c61f7747ff6","scrolled":true,"trusted":true},"cell_type":"code","source":"for ((train_idx, valid_idx), (train_f_idx, valid_f_idx)) in zip(splits, features_split):\n        print(train_idx)\n        print(valid_idx)\n        print(train_f_idx)\n        print(valid_f_idx)\n        X_train = train_X[train_idx]\n        y_train = train_y[train_idx]\n        X_val = train_X[valid_idx]\n        y_val = train_y[valid_idx]\n        Xfeaturestrain = np.array(x_features_train)[train_f_idx]\n        Xfeaturesval = np.array(x_features_train)[valid_f_idx]\n        model = model_gru_cap(embedding_matrix)\n        pred_val_y, pred_test_y, best_score = train_pred(model, X_train, y_train, X_val, y_val, Xfeaturestrain, Xfeaturesval, epochs = 6, callback = [clr,])\n        print(pred_val_y.shape)\n        print(pred_test_y.shape)\n        train_meta[valid_idx] = pred_val_y.reshape(-1)\n        test_meta += pred_test_y.reshape(-1) / len(splits)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"b45a347fe270d9ca5de985ecfa8cfd9c4896b711","trusted":true},"cell_type":"code","source":"threshold = threshold_search(train_y, train_meta)\nthreshold","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"af32f47e063fe4a2a74d754624173f8d43ccb256","trusted":true},"cell_type":"code","source":"f1_score(y_true=train_y, y_pred=train_meta > threshold[\"threshold\"])","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"e49910fd9c0ddf1244aaf08f1b2046a04c865f61","trusted":true},"cell_type":"code","source":"sub = pd.read_csv('../input/sample_submission.csv')\nsub.prediction = (test_meta > threshold[\"threshold\"]).astype(int)\nsub.to_csv(\"submission.csv\", index=False)","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}