{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nfrom tqdm import tqdm\ntqdm.pandas()\n%matplotlib inline\n\nfrom plotly import tools\nimport plotly.offline as py\npy.init_notebook_mode(connected=True)\nimport plotly.graph_objs as go\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\nimport math\nfrom sklearn.model_selection import train_test_split\nfrom sklearn import metrics\nimport gensim.models.keyedvectors as word2vec\nfrom keras.preprocessing.text import Tokenizer\nfrom keras.preprocessing.sequence import pad_sequences\nfrom keras.layers import Dense, Input, LSTM, Embedding, Dropout, Activation, CuDNNGRU, Conv1D, CuDNNLSTM, concatenate\nfrom keras.layers import Bidirectional, GlobalMaxPool1D, Dropout, SpatialDropout1D, GlobalAveragePooling1D, GlobalMaxPooling1D\nfrom keras.models import Model\nfrom keras import initializers, regularizers, constraints, optimizers, layers\nimport os\nprint(os.listdir(\"../input\"))\nimport gensim.models.keyedvectors as word2vec\nimport gc\n# Any results you write to the current directory are saved as output.","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"train=pd.read_csv(\"../input/train.csv\")\ntest=pd.read_csv(\"../input/test.csv\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"cec9fbff21fd62409f9e9e634fc9fd8106389db4"},"cell_type":"code","source":"train.head(10)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"e39a0f37a860963b7f1ed357955ae98adff2023e"},"cell_type":"markdown","source":"Counting the frequency of words in our data"},{"metadata":{"trusted":true,"_uuid":"ac1722621396dbeec6fd965de0b25ea6796fa35f"},"cell_type":"code","source":"def build_vocab(sentences,verbose=True):\n    vocab={}\n    \n    for sentence in tqdm(sentences,disable=(not verbose)):\n        for word in sentence:\n            try:\n                vocab[word] += 1\n            except KeyError:\n                vocab[word] = 1\n    return vocab\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"scrolled":true,"_uuid":"ad460da4616aec9b627c098aeea2f952275d5503"},"cell_type":"code","source":"sentences = train[\"question_text\"].progress_apply(lambda x: x.split()).values\n\nvocab = build_vocab(sentences)\nprint({k: vocab[k] for k in list(vocab)[:5]})","execution_count":null,"outputs":[]},{"metadata":{"_kg_hide-output":true,"_uuid":"72ab1af9d3c21160db24a9d496a30dede85d0ddc"},"cell_type":"markdown","source":"Now importing the embeddings and checking the percentage of words in data thats are in embeddings."},{"metadata":{"trusted":true,"_uuid":"97bb9b89c33297f6e9a65e9c3ce2fadcefae44b1"},"cell_type":"code","source":"from gensim.models import KeyedVectors\nnews_path = '../input/embeddings/GoogleNews-vectors-negative300/GoogleNews-vectors-negative300.bin'\nword2vecDict= KeyedVectors.load_word2vec_format(news_path, binary=True)\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"0b130f19d450d81be6938c6954357963eaf5e3e1"},"cell_type":"code","source":"import operator \n\ndef check_coverage(vocab,word2vecDict):\n    a = {}\n    oov = {}\n    k = 0\n    i = 0\n    for word in tqdm(vocab):\n        try:\n            a[word] = word2vecDict[word]\n            k += vocab[word]\n        except:\n\n            oov[word] = vocab[word]\n            i += vocab[word]\n            pass\n\n    print('Found embeddings for {:.2%} of vocab'.format(len(a) / len(vocab)))\n    print('Found embeddings for  {:.2%} of all text'.format(k / (k + i)))\n    sorted_x = sorted(oov.items(), key=operator.itemgetter(1))[::-1]\n\n    return sorted_x","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a0b8cbd0c10c4e00ca589cadd6f62b3dc04a2817"},"cell_type":"code","source":"oov = check_coverage(vocab,word2vecDict)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"fac8f64f164af8c6cf66a36c0dc629c3887ff4c0"},"cell_type":"markdown","source":"From this we can clearly see that only 24.3% of the data words are present in the embeddings. This shows there is a need for preprocessing the data to eleminate the puncuation marks, numbers, spaces. We go along the steps to clean the data. "},{"metadata":{"trusted":true,"_uuid":"cd066d94ea728eb2c52bc17a3d6e0e0fdb3d4164"},"cell_type":"markdown","source":"Lets look at the data words in data"},{"metadata":{"trusted":true,"_uuid":"d5bf3fa6ef528988bc42943f07240c97c11ab5b6"},"cell_type":"code","source":"oov[:15]","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"814500427d183fcd34d63f82e6766887941dc266"},"cell_type":"markdown","source":"Now punctuation are the reson for less percentage of words in embeddings. Lets remove these punctuations."},{"metadata":{"trusted":true,"_uuid":"326beacd7c55a0208d18dabb4290d0647c51befa"},"cell_type":"code","source":"def clean_text(x):\n    x = str(x)\n    for punct in \"/-'\":\n        x = x.replace(punct, ' ')\n    for punct in '&':\n        x = x.replace(punct, f' {punct} ')\n    for punct in '?!.,\"#$%\\'()*+-/:;<=>@[\\\\]^_`{|}~' + '“”’':\n        x = x.replace(punct, '')\n    return x\n                    ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"4a929fe61b6d6283441647f61136d821c9ab6326"},"cell_type":"code","source":"train_df[\"question_text\"] = train[\"question_text\"].progress_apply(lambda x: clean_text(x))\ntest_df[\"question_text\"] = test[\"question_text\"].progress_apply(lambda x: clean_text(x))\nsentences = train[\"question_text\"].apply(lambda x: x.split())\nvocab = build_vocab(sentences)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"2d69adcac61c32c6f04917a1fb2f24b1246ec399"},"cell_type":"code","source":"oov = check_coverage(vocab,word2vecDict)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b494285c39255e1ad49be0c24c6876bf9a9b135a"},"cell_type":"code","source":"oov[:16]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"79a9776671ea5d27ebef77c44fd8d7d1e9beeee3"},"cell_type":"code","source":"for i in range(10):\n    print(word2vecDict.index2entity[i])","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"0107de56c9b3d9335babbb9d5a0aa9e73807d9cd"},"cell_type":"markdown","source":"Looks like numbers are denoted with \"#\" so we need to reolace numbers with \"#\" as per the google embeddings."},{"metadata":{"trusted":true,"_uuid":"7267e71effdeea05c32a9854c767abfc272c9223"},"cell_type":"code","source":"import re\n\ndef clean_numbers(x):\n    x = re.sub('[0-9]{5,}', '#####', x)\n    x = re.sub('[0-9]{4}', '####', x)\n    x = re.sub('[0-9]{3}', '###', x)\n    x = re.sub('[0-9]{2}',\"##\",x)\n    return x\ntrain_df[\"question_text\"] = train[\"question_text\"].progress_apply(lambda x: clean_numbers(x))\ntest_df[\"question_text\"] = test[\"question_text\"].progress_apply(lambda x: clean_numbers(x))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"3f617a0e3643312812fda528a3becb2decc7f671"},"cell_type":"code","source":"sentences = train_df[\"question_text\"].progress_apply(lambda x: x.split())\nvocab = build_vocab(sentences)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c8d08d0ac99f465c8bfb6613d1bd19fde48739fa"},"cell_type":"code","source":"oov = check_coverage(vocab,word2vecDict)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"3557b280a69b79d790c930affb77a05151f57bce"},"cell_type":"code","source":"oov[:20]","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"47ba059d429db838cd51425c429c6ae5fc8c72f0"},"cell_type":"markdown","source":"Now considering for the missplled words and replacing with correct words and removing the words like \"a\",\"to\",\"of\",\"and\""},{"metadata":{"trusted":true,"_uuid":"e11d087cf038faa8feae7934c77971bc2ca210b5"},"cell_type":"code","source":"def _get_mispell(mispell_dict):\n    mispell_re = re.compile('(%s)' % '|'.join(mispell_dict.keys()))\n    return mispell_dict, mispell_re\n\n\nmispell_dict = {'colour':'color',\n                'centre':'center',\n                'didnt':'did not',\n                'doesnt':'does not',\n                'isnt':'is not',\n                'shouldnt':'should not',\n                'favourite':'favorite',\n                'travelling':'traveling',\n                'counselling':'counseling',\n                'theatre':'theater',\n                'cancelled':'canceled',\n                'labour':'labor',\n                'organisation':'organization',\n                'wwii':'world war 2',\n                'citicise':'criticize',\n                'instagram': 'social medium',\n                'whatsapp': 'social medium',\n                'snapchat': 'social medium'\n\n                }\nmispellings, mispellings_re = _get_mispell(mispell_dict)\n\ndef replace_typical_misspell(text):\n    def replace(match):\n        return mispellings[match.group(0)]\n\n    return mispellings_re.sub(replace, text)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"59fae47b20da3f83e3f3006eca0f8c5d272c8b8d"},"cell_type":"code","source":"train_df[\"question_text\"] = train[\"question_text\"].progress_apply(lambda x: replace_typical_misspell(x))\ntest_df[\"question_text\"] = test[\"question_text\"].progress_apply(lambda x: replace_typical_misspell(x))\nsentences = train[\"question_text\"].progress_apply(lambda x: x.split())\nto_remove = ['a','to','of','and']\nsentences = [[word for word in sentence if not word in to_remove] for sentence in tqdm(sentences)]\nvocab = build_vocab(sentences)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"7e8ea002248be170f10b834f87f95b9b6ee49ebe"},"cell_type":"code","source":"oov = check_coverage(vocab,word2vecDict)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"scrolled":false,"_uuid":"7cc2321ea9643903840f2d6599fb48e889abef4a"},"cell_type":"code","source":"oov[:20]\n","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"ba8a39cb92d9dbc39aa01f4493050ce4df7825f9"},"cell_type":"markdown","source":"\nNow the Data is processed and was almost 60% of the embedding.\nNow we move into data insights.\n"},{"metadata":{"trusted":true,"_uuid":"7a628e7c1bbca170656cdc9d8f4c3496b3381da4"},"cell_type":"code","source":"del(oov)\n\ngc.collect()\ntrain.head(20)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"9f268d2bb250d88b9e5cd9104190e16451190530"},"cell_type":"code","source":"embed_size = 300\nmaxlen = 200\nmax_features = 20000\ntokenizer = Tokenizer(num_words=max_features)\n\ntrain_x,val_x=train_test_split(train_df, test_size=0.1, random_state=2018)\ntrain_X=train_x[\"question_text\"].fillna(\"_na_\").values\nval_X=val_x[\"question_text\"].fillna(\"_na_\").values\ntest_X=test_df[\"question_text\"].fillna(\"_na_\").values\n\ntokenizer = Tokenizer(num_words=max_features)\ntokenizer.fit_on_texts(list(train_X))\ntrain_X = tokenizer.texts_to_sequences(train_X)\nval_X = tokenizer.texts_to_sequences(val_X)\ntest_X = tokenizer.texts_to_sequences(test_X)\n\n## Pad the sentences \ntrain_X = pad_sequences(train_X, maxlen=maxlen)\nval_X = pad_sequences(val_X, maxlen=maxlen)\ntest_X = pad_sequences(test_X, maxlen=maxlen)\n\n## Get the target values\ntrain_y = train_x['target'].values\nval_y = val_x['target'].values","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"0446fea5821a5eef4bd2b0cd6002597aae2c9f7c"},"cell_type":"code","source":"#word2vecDict = word2vec.KeyedVectors.load_word2vec_format(\"../input/embeddings/GoogleNews-vectors-negative300/GoogleNews-vectors-negative300.bin\", binary=True)\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f7c42b7d23abd7685e756d6c908f2c4797f65f8c"},"cell_type":"code","source":"embeddings_index = dict()\nfor word in word2vecDict.wv.vocab:\n    embeddings_index[word] = word2vecDict.word_vec(word)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"5cdf6a4f6725e480d38b79a9937dece1a488f77b"},"cell_type":"markdown","source":"Now going to model"},{"metadata":{"trusted":true,"_uuid":"f28fb607de528945792d19e4c91ce9f29991b6b2"},"cell_type":"code","source":"\nword_index = tokenizer.word_index\nnb_words = min(max_features, len(word_index))\nembedding_matrix = (np.random.rand(nb_words, embed_size) - 0.5) / 5.0\nfor word, i in word_index.items():\n    if i >= max_features: continue\n    if word in word2vecDict:\n        embedding_vector = word2vecDict.get_vector(word)\n        embedding_matrix[i] = embedding_vector\n        \ndel word2vecDict; gc.collect()   ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c0ee630d5912d5b3c04eb466d6ce7d5740ad2c13"},"cell_type":"code","source":"#inp = Input(shape=(maxlen,))\n#x = Embedding(max_features, embed_size, weights=[embedding_matrix])(inp)\n#x = Bidirectional(CuDNNGRU(64, return_sequences=True))(x)\n#x = GlobalMaxPool1D()(x)\n#x = Dense(16, activation=\"relu\")(x)\n#x = Dropout(0.1)(x)\n#x = Dense(1, activation=\"sigmoid\")(x)\n#model = Model(inputs=inp, outputs=x)\n#model.compile(loss='binary_crossentropy', optimizer='adam', metrics=['accuracy'])\nembed_size = 300 # how big is each word vector\nmax_features = 20000 # how many unique words to use (i.e num rows in embedding vector)\nmaxlen = 200 # max number of words in a question to use\n\nS_DROPOUT = 0.4\nDROPOUT = 0.1\n\n\ninp = Input(shape=(maxlen,))\nx = Embedding(max_features, embed_size , weights=[embedding_matrix_3])(inp)\nx = SpatialDropout1D(S_DROPOUT)(x)\nx = Bidirectional(CuDNNGRU(128, return_sequences=True))(x)\navg_pool = GlobalAveragePooling1D()(x)\nmax_pool = GlobalMaxPooling1D()(x)\nconc = concatenate([avg_pool, max_pool])\nx = Dense(16, activation=\"relu\")(conc)\nx = Dropout(DROPOUT)(x)\nx = Dense(1, activation=\"sigmoid\")(x)\nmodel = Model(inputs=inp, outputs=x)\nmodel.compile(loss='binary_crossentropy', optimizer='adam', metrics=['accuracy'])\n\nprint(model.summary())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c06ff0e34c43308aa25328e8896001b1df56801a"},"cell_type":"code","source":"EMBEDDING_FILE = '../input/embeddings/glove.840B.300d/glove.840B.300d.txt'\ndef get_coefs(word,*arr): return word, np.asarray(arr, dtype='float32')\nembeddings_index = dict(get_coefs(*o.split(\" \")) for o in open(EMBEDDING_FILE))\n\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"8d867b81de1ea6ed112a3ae4178da48ce08e8a2b"},"cell_type":"code","source":"embed_size = 300 \nmax_features = 50000 \nmaxlen = 100 \n\ntrain_x,val_x=train_test_split(train, test_size=0.1, random_state=42)\ntrain_X=train_x[\"question_text\"].fillna(\"_na_\").values\nval_X=val_x[\"question_text\"].fillna(\"_na_\").values\ntest_X=test[\"question_text\"].fillna(\"_na_\").values\ntokenizer = Tokenizer(num_words=max_features)\n\ntokenizer.fit_on_texts(list(train_X))\ntrain_X = tokenizer.texts_to_sequences(train_X)\nval_X = tokenizer.texts_to_sequences(val_X)\ntest_X = tokenizer.texts_to_sequences(test_X)\n\n## Pad the sentences \ntrain_X = pad_sequences(train_X, maxlen=maxlen)\nval_X = pad_sequences(val_X, maxlen=maxlen)\ntest_X = pad_sequences(test_X, maxlen=maxlen)\n\n## Get the target values\ntrain_y = train_x['target'].values\nval_y = val_x['target'].values\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"7c1fe794d8dd628fa1772b09cf81df54f5ee41c2"},"cell_type":"code","source":"\nall_embs = np.stack(embeddings_index.values())\nemb_mean,emb_std = all_embs.mean(), all_embs.std()\nembed_size = all_embs.shape[1]\n\nword_index = tokenizer.word_index\nnb_words = min(max_features, len(word_index))\nembedding_matrix = np.random.normal(emb_mean, emb_std, (nb_words, embed_size))\nfor word, i in word_index.items():\n    if i >= max_features: continue\n    embedding_vector = embeddings_index.get(word)\n    if embedding_vector is not None: embedding_matrix[i] = embedding_vector\n        \ninp = Input(shape=(maxlen,))\nx = Embedding(max_features, embed_size, weights=[embedding_matrix])(inp)\nx = Bidirectional(CuDNNGRU(64, return_sequences=True))(x)\nx = GlobalMaxPool1D()(x)\nx = Dense(16, activation=\"relu\")(x)\nx = Dropout(0.1)(x)\nx = Dense(1, activation=\"sigmoid\")(x)\nmodel = Model(inputs=inp, outputs=x)\nmodel.compile(loss='binary_crossentropy', optimizer='adam', metrics=['accuracy'])\nprint(model.summary())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"71332fe6481ac1c57952a67efed78806709418ae"},"cell_type":"code","source":"model.fit(train_X, train_y, batch_size=1024, epochs=2, validation_data=(val_X, val_y))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"4339e3878a57048fc56566b4a3e4e8275c1567c9"},"cell_type":"code","source":"pred_test_y = model.predict([val_X], batch_size=512, verbose=1)\nfor thresh in np.arange(0.1, 0.501, 0.01):\n    thresh = np.round(thresh, 2)\n    print(\"F1 score at threshold {0} is {1}\".format(thresh, metrics.f1_score(val_y, (pred_test_y>thresh).astype(int))))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c0becafea33afb79994aeb84563529a9283cd3f3"},"cell_type":"code","source":"pred_y = model.predict([test_X], batch_size=512, verbose=1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"6bc2785acb0616f08cacc0efd30b9f233288ef18"},"cell_type":"code","source":"\npred_y = (pred_y>0.35).astype(int)\nout= pd.DataFrame({\"qid\":test[\"qid\"].values})\nout['prediction'] = pred_y\nout.to_csv(\"submission.csv\", index=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"0331079aa402172dac7b14e157d12ac127fc61c8"},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"1a6d1a756759bf353c0d8bfd063ca31595085350"},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"6019612a67b63ee0f26e8392734df7480e04428e"},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}