{"cells":[{"metadata":{"_uuid":"eb721e8c304d6beeaecf948903154e25bddbb53a"},"cell_type":"markdown","source":"# Stacking CNN + Google News Vectors\nFirst, thanks to [Dieter](https://www.kaggle.com/christofhenkel/how-to-preprocessing-when-using-embeddings) for his preprocessing with pre-trained embeddings tutorial.\n\nThis kernel was also built with the help of [Machine Learning Mastery's](https://machinelearningmastery.com/stacking-ensemble-for-deep-learning-neural-networks/) tutorial on developing stacking ensembles for deep learning.\n\nStacking can improve model performance by combining predictions from multiple sub-models.  It works by taking the outputs of the sub-models as imputs and attempting to learn how to best combine them to make an improved output prediction."},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nfrom tqdm import tqdm\nimport re\n\ntqdm.pandas()\n\nimport warnings\nwarnings.filterwarnings('ignore')","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"train = pd.read_csv(\"../input/train.csv\")\ntest = pd.read_csv(\"../input/test.csv\")\nprint(\"Train shape : \",train.shape)\nprint(\"Test shape : \",test.shape)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"5f81231d640d4ec4916570a19a1f1752aa9732de"},"cell_type":"code","source":"# text preprocessing\ncontractions = {\n\"ain't\": \"is not\",\n\"aren't\": \"are not\",\n\"can't\": \"cannot\",\n\"can't've\": \"cannot have\",\n\"'cause\": \"because\",\n\"could've\": \"could have\",\n\"couldn't\": \"could not\",\n\"couldn't've\": \"could not have\",\n\"didn't\": \"did not\",\n\"doesn't\": \"does not\",\n\"don't\": \"do not\",\n\"hadn't\": \"had not\",\n\"hadn't've\": \"had not have\",\n\"hasn't\": \"has not\",\n\"haven't\": \"have not\",\n\"he'd\": \"he would\",\n\"he'd've\": \"he would have\",\n\"he'll\": \"he will\",\n\"he'll've\": \"he he will have\",\n\"he's\": \"he is\",\n\"how'd\": \"how did\",\n\"how'd'y\": \"how do you\",\n\"how'll\": \"how will\",\n\"how's\": \"how is\",\n\"I'd\": \"I would\",\n\"I'd've\": \"I would have\",\n\"I'll\": \"I will\",\n\"I'll've\": \"I will have\",\n\"I'm\": \"I am\",\n\"I've\": \"I have\",\n\"i'd\": \"i would\",\n\"i'd've\": \"i would have\",\n\"i'll\": \"i will\",\n\"i'll've\": \"i will have\",\n\"i'm\": \"i am\",\n\"i've\": \"i have\",\n\"isn't\": \"is not\",\n\"it'd\": \"it would\",\n\"it'd've\": \"it would have\",\n\"it'll\": \"it will\",\n\"it'll've\": \"it will have\",\n\"it's\": \"it is\",\n\"let's\": \"let us\",\n\"ma'am\": \"madam\",\n\"mayn't\": \"may not\",\n\"might've\": \"might have\",\n\"mightn't\": \"might not\",\n\"mightn't've\": \"might not have\",\n\"must've\": \"must have\",\n\"mustn't\": \"must not\",\n\"mustn't've\": \"must not have\",\n\"needn't\": \"need not\",\n\"needn't've\": \"need not have\",\n\"o'clock\": \"of the clock\",\n\"oughtn't\": \"ought not\",\n\"oughtn't've\": \"ought not have\",\n\"shan't\": \"shall not\",\n\"sha'n't\": \"shall not\",\n\"shan't've\": \"shall not have\",\n\"she'd\": \"she would\",\n\"she'd've\": \"she would have\",\n\"she'll\": \"she will\",\n\"she'll've\": \"she will have\",\n\"she's\": \"she is\",\n\"should've\": \"should have\",\n\"shouldn't\": \"should not\",\n\"shouldn't've\": \"should not have\",\n\"so've\": \"so have\",\n\"so's\": \"so as\",\n\"that'd\": \"that would\",\n\"that'd've\": \"that would have\",\n\"that's\": \"that is\",\n\"there'd\": \"there would\",\n\"there'd've\": \"there would have\",\n\"there's\": \"there is\",\n\"they'd\": \"they would\",\n\"they'd've\": \"they would have\",\n\"they'll\": \"they will\",\n\"they'll've\": \"they will have\",\n\"they're\": \"they are\",\n\"they've\": \"they have\",\n\"to've\": \"to have\",\n\"wasn't\": \"was not\",\n\"we'd\": \"we would\",\n\"we'd've\": \"we would have\",\n\"we'll\": \"we will\",\n\"we'll've\": \"we will have\",\n\"we're\": \"we are\",\n\"we've\": \"we have\",\n\"weren't\": \"were not\",\n\"what'll\": \"what will\",\n\"what'll've\": \"what will have\",\n\"what're\": \"what are\",\n\"what's\": \"what is\",\n\"what've\": \"what have\",\n\"when's\": \"when is\",\n\"when've\": \"when have\",\n\"where'd\": \"where did\",\n\"where's\": \"where is\",\n\"where've\": \"where have\",\n\"who'll\": \"who will\",\n\"who'll've\": \"who will have\",\n\"who's\": \"who is\",\n\"who've\": \"who have\",\n\"why's\": \"why is\",\n\"why've\": \"why have\",\n\"will've\": \"will have\",\n\"won't\": \"will not\",\n\"won't've\": \"will not have\",\n\"would've\": \"would have\",\n\"wouldn't\": \"would not\",\n\"wouldn't've\": \"would not have\",\n\"y'all\": \"you all\",\n\"y'all'd\": \"you all would\",\n\"y'all'd've\": \"you all would have\",\n\"y'all're\": \"you all are\",\n\"y'all've\": \"you all have\",\n\"you'd\": \"you would\",\n\"you'd've\": \"you would have\",\n\"you'll\": \"you will\",\n\"you'll've\": \"you will have\",\n\"you're\": \"you are\",\n\"you've\": \"you have\"\n}\n\nc_re = re.compile('(%s)' % '|'.join(contractions.keys()))\n\ndef expandContractions(text, c_re=c_re):\n    def replace(match):\n        return contractions[match.group(0)]\n    return c_re.sub(replace, text)\n\nfrom gensim.parsing.preprocessing import preprocess_string\nfrom gensim.parsing.preprocessing import strip_tags, strip_punctuation, strip_numeric\nfrom gensim.parsing.preprocessing import strip_multiple_whitespaces, strip_non_alphanum, remove_stopwords, strip_short\n\nCUSTOM_FILTERS = [#lambda x: x.lower(), #lowercase\n                  strip_tags, # remove html tags\n                  #strip_punctuation, # replace punctuation with space\n                  strip_multiple_whitespaces,# remove repeating whitespaces\n                  strip_non_alphanum, # remove non-alphanumeric characters\n                  #strip_numeric, # remove numbers\n                  #remove_stopwords,# remove stopwords\n                  strip_short # remove words less than minsize=3 characters long\n                 ]\ndef gensim_preprocess(docs):\n    docs = [expandContractions(doc) for doc in docs]\n    docs = [preprocess_string(text, CUSTOM_FILTERS) for text in docs]\n    docs = [' '.join(text) for text in docs]\n    return pd.Series(docs)\n\ntrain_clean = gensim_preprocess(train.question_text)\n\ngensim_preprocess(train.question_text.iloc[10:15])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"80515b1714cd847c2f5a90f501d5f10290dd4af8"},"cell_type":"code","source":"# creating vocab from train dataframe\nfrom collections import Counter\nvocab = Counter()\n\ntexts = ' '.join(train_clean).split()\nvocab.update(texts)\n\nprint(len(vocab))\nprint(vocab.most_common(50))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e01724db7dc07927f0c7246edc08bfdea1c54c0e"},"cell_type":"code","source":"# load google news vectors\nfrom gensim.models import KeyedVectors\nnews_path = '../input/embeddings/GoogleNews-vectors-negative300/GoogleNews-vectors-negative300.bin'\nembeddings_index = KeyedVectors.load_word2vec_format(news_path, binary=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c5a8d54d8670ea0e54726ad09d28e6090a0cd5f6"},"cell_type":"code","source":"# function to check coverage of embedding vs train vocabulary\nimport operator \n\ndef check_coverage(vocab,embeddings_index):\n    a = {}\n    oov = {}\n    k = 0\n    i = 0\n    for word in tqdm(vocab):\n        try:\n            a[word] = embeddings_index[word]\n            k += vocab[word]\n        except:\n\n            oov[word] = vocab[word]\n            i += vocab[word]\n            pass\n\n    print('Found embeddings for {:.2%} of vocab'.format(len(a) / len(vocab)))\n    print('Found embeddings for  {:.2%} of all text'.format(k / (k + i)))\n    sorted_x = sorted(oov.items(), key=operator.itemgetter(1))[::-1]\n\n    return sorted_x","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c050f63f248cef758cd2328e044c8ee9160343a8"},"cell_type":"code","source":"# function to correct misspellings and out of vocab words\ndef _get_mispell(mispell_dict):\n    mispell_re = re.compile('(%s)' % '|'.join(mispell_dict.keys()))\n    return mispell_dict, mispell_re\n\n\nmispell_dict = {'colour':'color',\n                'centre':'center',\n                'favourite':'favorite',\n                'travelling':'traveling',\n                'counselling':'counseling',\n                'theatre':'theater',\n                'cancelled':'canceled',\n                'labour':'labor',\n                'organisation':'organization',\n                'wwii':'world war 2',\n                'citicise':'criticize',\n                'instagram': 'social medium',\n                'whatsapp': 'social medium',\n                'snapchat': 'social medium',\n                'Snapchat': 'social medium',\n                'quora': 'social medium',\n                'Quora': 'social medium',\n                'mediumns': 'mediums',\n                'bitcoin': 'currency',\n                'cryptocurrency': 'currency',\n                'upsc': 'union public service commission',\n                'mbbs': 'bachelor medicine',\n                'ece': 'educational credential evaluators',\n                'aiims': 'all india institute medical science',\n                'iim': 'india institute management',\n                'sbi': 'state bank india',\n                'blockchain': 'crytography',\n                'and': '',\n                'reducational':'educational',\n                'neducational':'educational',\n                'greeducational': 'greed educational',\n                'pieducational': 'educational',\n                'deducational': 'educational',\n                'Quorans': 'Quoran'   \n                }\nmispellings, mispellings_re = _get_mispell(mispell_dict)\n\ndef replace_typical_misspell(text):\n    def replace(match):\n        return mispellings[match.group(0)]\n\n    return mispellings_re.sub(replace, text)\n\n# replace numbers > 9 with #### to match embedding\ndef clean_numbers(x):\n\n    x = re.sub('[0-9]{5,}', '#####', x)\n    x = re.sub('[0-9]{4}', '####', x)\n    x = re.sub('[0-9]{3}', '###', x)\n    x = re.sub('[0-9]{2}', '##', x)\n    return x\n\ntrain_clean = train_clean.apply(lambda x: replace_typical_misspell(x))\ntrain_clean = train_clean.apply(lambda x: clean_numbers(x))\n\nvocab = Counter()\ntexts = ' '.join(train_clean).split()\nvocab.update(texts)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e8ff208ed3b95f97cdb89c45b7071f95d1e32a5a"},"cell_type":"code","source":"# check out of vocab words again\noov = check_coverage(vocab,embeddings_index)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f4800e22d86ffd281b0fcfdc64ed165fac34f222"},"cell_type":"code","source":"# view top 20 oov words\noov[:20]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"5db29514870b62a7649595b54f44f30ed9009d2e"},"cell_type":"code","source":"# clean up our vocab\n# keep tokens with a min occurrence\nmin_occurrence = 5\ntokens = [k for k,c in vocab.items() if c >= min_occurrence]\nprint(len(tokens))\n\nvocab = set((' '.join(tokens)).split())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"87eab7bdd710519890330e59ac1a84cb56f8fe07"},"cell_type":"code","source":"from keras.preprocessing.text import Tokenizer\nfrom keras.preprocessing.sequence import pad_sequences\nfrom keras.utils.vis_utils import plot_model\nfrom keras.models import Sequential\nfrom keras.layers import Dense\nfrom keras.layers import Flatten\nfrom keras.layers import Embedding\nfrom keras.layers.convolutional import Conv1D\nfrom keras.layers.convolutional import MaxPooling1D\n\n\n# fit a tokenizer using keras\ndef create_tokenizer(text):\n    tokenizer = Tokenizer()\n    tokenizer.fit_on_texts(text)\n    return tokenizer\ntokenizer = create_tokenizer(train_clean)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"bb7e48a6327403ae338bb97baed81ec9bd2f8f4d"},"cell_type":"code","source":"def encode_docs(tokenizer, max_length, docs):\n    # integer encode\n    encoded = tokenizer.texts_to_sequences(docs)\n    # pad sequences\n    padded = pad_sequences(encoded, maxlen = max_length, padding='post')\n    return padded","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e618afdaa9d4a72227c05341b0cc099de96d40ef"},"cell_type":"code","source":"vocab_size = len(tokenizer.word_index) + 1\nprint('Vocab Size: ', vocab_size)\nmax_length = max([len(s.split()) for s in train_clean])\nprint('Max Length: ', max_length)\nX_train = encode_docs(tokenizer, max_length, train_clean)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"3596acf4964083f50c3323b1b3113ef861a5deda"},"cell_type":"code","source":"from keras.initializers import Constant\ntype(embeddings_index.vocab)\nEMBEDDING_DIM = 300\nMAX_NUM_WORDS = 30000\nword_index = tokenizer.word_index\nnum_words = vocab_size\nembedding_matrix = np.zeros((vocab_size, EMBEDDING_DIM))\nfor word, i in word_index.items():\n    if i > MAX_NUM_WORDS:\n        continue\n    try:\n        embedding_vector = embeddings_index.get_vector(word)\n    \n        # words not found in embedding index will be all-zeros.\n        embedding_matrix[i] = embedding_vector\n    except (KeyError):\n        continue\n        \n# load pre-trained word embeddings into an Embedding layer\n# note that we set trainable = False so as to keep the embeddings fixed\nembedding_layer = Embedding(num_words,\n                            EMBEDDING_DIM,\n                            embeddings_initializer=Constant(embedding_matrix),\n                            input_length=max_length,\n                            trainable=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"43bcbf48e6e61a2ab5a923537d31012cd8743b3e"},"cell_type":"code","source":"# define and fit model on dataset\ny_train = train.target\n\ndef define_model(vocab_size, max_length):\n    #define model\n    model = Sequential()\n    model.add(embedding_layer)\n    model.add(Conv1D(filters=32, kernel_size=8, activation='relu'))\n    model.add(MaxPooling1D(pool_size=2))\n    model.add(Flatten())\n    model.add(Dense(10, activation='relu'))\n    model.add(Dense(1, activation='sigmoid'))\n    # compile network\n    model.compile(loss='binary_crossentropy', optimizer='adam', metrics=['accuracy'])\n    return model\n\nmodel = define_model(vocab_size, max_length)\n\ndef fit_model(X_train, y_train, epochs = 1):\n    model.fit(X_train,\n              y_train,\n              epochs=epochs,\n              verbose=1,\n              shuffle=True,\n              validation_split=0.1,\n              class_weight={1:0.6, 0:0.4})\n    return model","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"220be115722f5584a2b141d6851cda2cc3fde600"},"cell_type":"code","source":"# create submodels\nn_members = 5\nmembers = []\nfor i in range(n_members):\n    # fit model\n    m = fit_model(X_train, y_train, epochs = (i+1))\n    members.append(m)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"474ea3743b0529cac730f2c4ca487a2eb6e4a819"},"cell_type":"code","source":"# create stacked model input dataset as outputs from the ensemble\ndef stacked_dataset(members, X_train):\n    stackX = None\n    for model in members:\n        # generate class probability prediction\n        yhat = model.predict_proba(X_train, verbose=0)\n        # stack predictions into [rows, predictions]\n        if stackX == None:\n            stackX = yhat\n        else:\n            stackX = dstack((stackX, yhat))\n            print(stackX.shape)\n        # flatten predictions to [rows, predictions]\n        stackX = stackX.reshape((stackX.shape[0], stackX.shape[1]))\n        return stackX","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f07518ff5c1281b5ce95073d0d0252e43054de85"},"cell_type":"code","source":"# separate stacking model\nfrom sklearn.linear_model import LogisticRegression\n\n# fit model based on outputs from ensemble members\ndef fit_stacked_model(members, inputX, inputy):\n    # create dataset using ensemble\n    stackedX = stacked_dataset(members, inputX)\n    # fit standalone model\n    model = LogisticRegression()\n    model.fit(stackedX, inputy)\n    return model\n\n# make a prediction with the stacked model\ndef stacked_prediction(members, model, inputX):\n    # create dataset using ensemble\n    stackedX = stacked_dataset(members, inputX)\n    # make a prediction\n    yhat = model.predict(stackedX)\n    return yhat","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"456c2f9a5076560db4f9a961e822e0b3ce98ff85"},"cell_type":"code","source":"model = fit_stacked_model(members, X_train, y_train)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f46f8ac5db17c92553229c5802ca673db7b36001"},"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nX_train, X_test, y_train, y_test = train_test_split(X_train,\n                                                y_train, test_size=0.2)\n\n# evaluate model on test set\ny_ = stacked_prediction(members, model, X_test)\n\nfrom sklearn.metrics import f1_score\nfrom sklearn.metrics import classification_report\n\nprint(f1_score(y_test, y_))\nprint(classification_report(y_test, y_))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"9c97ebfdd9210d7a8e6bdeafe66de40d0c53e100"},"cell_type":"code","source":"# prepare test data\ntest_clean = gensim_preprocess(test.question_text)\npred = encode_docs(tokenizer, max_length, test_clean)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"7220a661a5d08d38eefed111d040cd71f74035e2"},"cell_type":"code","source":"# predict on test data\nprediction = stacked_prediction(members, model, pred)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"78f83df345d3896addf57dedc4b4a7d8b1422657"},"cell_type":"code","source":"submission = pd.DataFrame({'qid':test.qid, 'prediction':prediction})\nsubmission.to_csv('submission.csv', index=False)\nsubmission.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b28053804bb66833d65f9fdc9241d17c7763fc0b"},"cell_type":"markdown","source":"This kernel did not outperform our previous version without stacking."}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}