{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import os\nimport time\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nfrom tqdm import tqdm\nimport math\nfrom sklearn.model_selection import train_test_split\nfrom sklearn import metrics\n\nfrom keras.preprocessing.text import Tokenizer\nfrom keras.preprocessing.sequence import pad_sequences\nfrom keras.layers import Dense, Input, LSTM, Embedding, Dropout, Activation, CuDNNGRU, Conv1D\nfrom keras.layers import Bidirectional, GlobalMaxPool1D\nfrom keras.models import Model\nfrom keras import initializers, regularizers, constraints, optimizers, layers","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"aa382b94a47fc6b0489bbb960b8e084ab872f7e1"},"cell_type":"code","source":"import nltk\nimport string\nfrom nltk.corpus import stopwords\nfrom nltk.stem.snowball import SnowballStemmer\nimport re","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"train_df = pd.read_csv(\"../input/train.csv\")\ntest_df = pd.read_csv(\"../input/test.csv\")\nprint(\"Train shape : \",train_df.shape)\nprint(\"Test shape : \",test_df.shape)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e6a637e4e0d6e8370d820d05567a9b895bea8024"},"cell_type":"code","source":"### Text Normalizing function\ndef clean_text(text): \n    ## Remove puncuation\n    text = text.translate(string.punctuation)\n    \n    ## Convert words to lower case and split them\n    text = text.lower().split()\n    \n    ## Remove stop words\n    stops = set(stopwords.words(\"english\"))\n    text = [w for w in text if not w in stops and len(w) >= 3]\n    \n    text = \" \".join(text)\n    ## Clean the text\n    text = re.sub(r\"[^A-Za-z0-9^,!.\\/'+-=]\", \" \", text)\n    text = re.sub(r\"what's\", \"what is \", text)\n    text = re.sub(r\"\\'s\", \" \", text)\n    text = re.sub(r\"\\'ve\", \" have \", text)\n    text = re.sub(r\"n't\", \" not \", text)\n    text = re.sub(r\"i'm\", \"i am \", text)\n    text = re.sub(r\"\\'re\", \" are \", text)\n    text = re.sub(r\"\\'d\", \" would \", text)\n    text = re.sub(r\"\\'ll\", \" will \", text)\n    text = re.sub(r\",\", \" \", text)\n    text = re.sub(r\"\\.\", \" \", text)\n    text = re.sub(r\"!\", \" ! \", text)\n    text = re.sub(r\"\\/\", \" \", text)\n    text = re.sub(r\"\\^\", \" ^ \", text)\n    text = re.sub(r\"\\+\", \" + \", text)\n    text = re.sub(r\"\\-\", \" - \", text)\n    text = re.sub(r\"\\=\", \" = \", text)\n    text = re.sub(r\"'\", \" \", text)\n    text = re.sub(r\"(\\d+)(k)\", r\"\\g<1>000\", text)\n    text = re.sub(r\":\", \" : \", text)\n    text = re.sub(r\" e g \", \" eg \", text)\n    text = re.sub(r\" b g \", \" bg \", text)\n    text = re.sub(r\" u s \", \" american \", text)\n    text = re.sub(r\"\\0s\", \"0\", text)\n    text = re.sub(r\" 9 11 \", \"911\", text)\n    text = re.sub(r\"e - mail\", \"email\", text)\n    text = re.sub(r\"j k\", \"jk\", text)\n    text = re.sub(r\"\\s{2,}\", \" \", text)\n    ## Stemming\n    text = text.split()\n    stemmer = SnowballStemmer('english')\n    stemmed_words = [stemmer.stem(word) for word in text]\n    text = \" \".join(stemmed_words)\n    return text\nprint(\"Clean Text Function defined\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d98b0b57cd2d6bde35392079a13000897de83d3a"},"cell_type":"code","source":"print(\"Start Cleaning Data texts\")\ntrain_df[\"question_text\"] = train_df[\"question_text\"].map(lambda x: clean_text(x))\ntest_df[\"question_text\"] = test_df[\"question_text\"].map(lambda x: clean_text(x))\nprint(\"Data Questions Text Cleaned\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"9fd6f59407ded15a2b1bd3afc94b05cc3b910a81"},"cell_type":"code","source":"## split to train and val\ntrain_df, val_df = train_test_split(train_df, test_size=0.1, random_state=2018)\n\n## some config values \nembed_size = 300 # how big is each word vector\nmax_features = 50000 # how many unique words to use (i.e num rows in embedding vector)\nmaxlen = 100 # max number of words in a question to use\n\n## fill up the missing values\ntrain_X = train_df[\"question_text\"].fillna(\"_na_\").values\nval_X = val_df[\"question_text\"].fillna(\"_na_\").values\ntest_X = test_df[\"question_text\"].fillna(\"_na_\").values\n\n## Tokenize the sentences\ntokenizer = Tokenizer(num_words=max_features)\ntokenizer.fit_on_texts(list(train_X))\ntrain_X = tokenizer.texts_to_sequences(train_X)\nval_X = tokenizer.texts_to_sequences(val_X)\ntest_X = tokenizer.texts_to_sequences(test_X)\n\n## Pad the sentences \ntrain_X = pad_sequences(train_X, maxlen=maxlen)\nval_X = pad_sequences(val_X, maxlen=maxlen)\ntest_X = pad_sequences(test_X, maxlen=maxlen)\n\n## Get the target values\ntrain_y = train_df['target'].values\nval_y = val_df['target'].values","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"17cffb42b10c47510463829ae1e862278947b549"},"cell_type":"code","source":"print(\"Creating Model by GLOVE Embeddings\")\nEMBEDDING_FILE = '../input/embeddings/glove.840B.300d/glove.840B.300d.txt'\ndef get_coefs(word,*arr): return word, np.asarray(arr, dtype='float32')\nembeddings_index = dict(get_coefs(*o.split(\" \")) for o in open(EMBEDDING_FILE))\n\nall_embs = np.stack(embeddings_index.values())\nemb_mean,emb_std = all_embs.mean(), all_embs.std()\nembed_size = all_embs.shape[1]\n\nword_index = tokenizer.word_index\nnb_words = min(max_features, len(word_index))\nembedding_matrix = np.random.normal(emb_mean, emb_std, (nb_words, embed_size))\nfor word, i in word_index.items():\n    if i >= max_features: continue\n    embedding_vector = embeddings_index.get(word)\n    if embedding_vector is not None: embedding_matrix[i] = embedding_vector\n        \ninp = Input(shape=(maxlen,))\nx = Embedding(max_features, embed_size, weights=[embedding_matrix])(inp)\nx = Bidirectional(CuDNNGRU(64, return_sequences=True))(x)\nx = GlobalMaxPool1D()(x)\nx = Dense(16, activation=\"relu\")(x)\nx = Dropout(0.1)(x)\nx = Dense(1, activation=\"sigmoid\")(x)\nmodel = Model(inputs=inp, outputs=x)\nmodel.compile(loss='binary_crossentropy', optimizer='adam', metrics=['accuracy'])\nprint(model.summary())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c8f7f7b09ff9ec6aa9b5b77207f8159d972a76b8"},"cell_type":"code","source":"model.fit(train_X, train_y, batch_size=512, epochs=3, validation_data=(val_X, val_y))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c3be8405f8e74e31f709d27eb750db19976e28f4"},"cell_type":"code","source":"print(\"Starting Prediction by GLOVE Embeddings for VAL\")\npred_glove_val_y = model.predict([val_X], batch_size=1024, verbose=1)\nbest_thresh = 0.5\nbest_score = 0.0\nfor thresh in np.arange(0.1, 0.501, 0.01):\n    thresh = np.round(thresh, 2)\n    score = metrics.f1_score(val_y, (pred_glove_val_y>thresh).astype(int))\n    print(\"F1 score at threshold {0} is {1}\".format(thresh, metrics.f1_score(val_y, (pred_glove_val_y>thresh).astype(int))))\n    if score > best_score:\n        best_thresh = thresh\n        best_score = score","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"67219f7774de43644b03be5c9e09342b55a3a235"},"cell_type":"code","source":"print()\nprint(\"Best Tresh for GLOVE is {0} at score {1}\".format(best_thresh,best_score))\nprint()\npred_glove_val_y = (pred_glove_val_y>best_thresh)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a81d213948ed60f7d1a27752b39a5155b395e74e"},"cell_type":"code","source":"print(\"Prediction for test based on GLOVE Embeddings\")\npred_glove_test_y = model.predict([test_X], batch_size=1024, verbose=1)\npred_glove_test_y = (pred_glove_test_y>best_thresh)\nprint(\"Glove Emebddings Prediction Complete\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"9980ff803962c03ba5bafe2b45998fbb8bb7c451"},"cell_type":"code","source":"del word_index, embeddings_index, all_embs, embedding_matrix, model, inp, x\nimport gc; gc.collect()\ntime.sleep(10)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a7a4c86e0dad985c4af6a6c772025381d059011f"},"cell_type":"code","source":"print(\"Creating Model by Wikinews Embeddings\")\nEMBEDDING_FILE = '../input/embeddings/wiki-news-300d-1M/wiki-news-300d-1M.vec'\ndef get_coefs(word,*arr): return word, np.asarray(arr, dtype='float32')\nembeddings_index = dict(get_coefs(*o.split(\" \")) for o in open(EMBEDDING_FILE) if len(o)>100)\n\nall_embs = np.stack(embeddings_index.values())\nemb_mean,emb_std = all_embs.mean(), all_embs.std()\nembed_size = all_embs.shape[1]\n\nword_index = tokenizer.word_index\nnb_words = min(max_features, len(word_index))\nembedding_matrix = np.random.normal(emb_mean, emb_std, (nb_words, embed_size))\nfor word, i in word_index.items():\n    if i >= max_features: continue\n    embedding_vector = embeddings_index.get(word)\n    if embedding_vector is not None: embedding_matrix[i] = embedding_vector\n        \ninp = Input(shape=(maxlen,))\nx = Embedding(max_features, embed_size, weights=[embedding_matrix])(inp)\nx = Bidirectional(CuDNNGRU(64, return_sequences=True))(x)\nx = GlobalMaxPool1D()(x)\nx = Dense(16, activation=\"relu\")(x)\nx = Dropout(0.1)(x)\nx = Dense(1, activation=\"sigmoid\")(x)\nmodel = Model(inputs=inp, outputs=x)\nmodel.compile(loss='binary_crossentropy', optimizer='adam', metrics=['accuracy'])\nprint(model.summary())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"4308c1b060c71428a2f89b37493112efa66484ed"},"cell_type":"code","source":"model.fit(train_X, train_y, batch_size=512, epochs=3, validation_data=(val_X, val_y))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"caa52c031f79cd6a21666ec32461e0257061b47f"},"cell_type":"code","source":"print(\"Starting Prediction by wikinews Embeddings for VAL\")\npred_fasttext_val_y = model.predict([val_X], batch_size=1024, verbose=1)\nbest_thresh = 0.5\nbest_score = 0.0\nfor thresh in np.arange(0.1, 0.501, 0.01):\n    thresh = np.round(thresh, 2)\n    score = metrics.f1_score(val_y, (pred_fasttext_val_y>thresh).astype(int))\n    print(\"F1 score at threshold {0} is {1}\".format(thresh, metrics.f1_score(val_y, (pred_fasttext_val_y>thresh).astype(int))))\n    if score > best_score:\n        best_thresh = thresh\n        best_score = score","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"3adfa1aefcec7386a3dcaec10d119394593bb770"},"cell_type":"code","source":"print()\nprint(\"Best Tresh for wikinews is {0} at score {1}\".format(best_thresh,best_score))\nprint()\npred_fasttext_val_y = (pred_fasttext_val_y>best_thresh)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"5c4f5d636ae29dbc66dfba9aca4e16472858fbb4"},"cell_type":"code","source":"print(\"Prediction for test based on wikinews Embeddings\")\npred_fasttext_test_y = model.predict([test_X], batch_size=1024, verbose=1)\npred_fasttext_test_y = (pred_fasttext_test_y>best_thresh)\nprint(\"Wikinews Emebddings Prediction Complete\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"bb6f3cb412e7510a04b1cfa40ba00f1c91e10e1c"},"cell_type":"code","source":"del word_index, embeddings_index, all_embs, embedding_matrix, model, inp, x\nimport gc; gc.collect()\ntime.sleep(10)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"189a352153ef70f1f1af30c021b36a0cb81106b0"},"cell_type":"code","source":"print(\"Creating Model by paragram embeddings\")\nEMBEDDING_FILE = '../input/embeddings/paragram_300_sl999/paragram_300_sl999.txt'\ndef get_coefs(word,*arr): return word, np.asarray(arr, dtype='float32')\nembeddings_index = dict(get_coefs(*o.split(\" \")) for o in open(EMBEDDING_FILE, encoding=\"utf8\", errors='ignore') if len(o)>100)\n\nall_embs = np.stack(embeddings_index.values())\nemb_mean,emb_std = all_embs.mean(), all_embs.std()\nembed_size = all_embs.shape[1]\n\nword_index = tokenizer.word_index\nnb_words = min(max_features, len(word_index))\nembedding_matrix = np.random.normal(emb_mean, emb_std, (nb_words, embed_size))\nfor word, i in word_index.items():\n    if i >= max_features: continue\n    embedding_vector = embeddings_index.get(word)\n    if embedding_vector is not None: embedding_matrix[i] = embedding_vector\n        \ninp = Input(shape=(maxlen,))\nx = Embedding(max_features, embed_size, weights=[embedding_matrix])(inp)\nx = Bidirectional(CuDNNGRU(64, return_sequences=True))(x)\nx = GlobalMaxPool1D()(x)\nx = Dense(16, activation=\"relu\")(x)\nx = Dropout(0.1)(x)\nx = Dense(1, activation=\"sigmoid\")(x)\nmodel = Model(inputs=inp, outputs=x)\nmodel.compile(loss='binary_crossentropy', optimizer='adam', metrics=['accuracy'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f0d757c0c6bf0845f99f4bea1e37dc10cf270029"},"cell_type":"code","source":"model.fit(train_X, train_y, batch_size=512, epochs=3, validation_data=(val_X, val_y))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"381ba87e674100f6775f49ce62a075ff57b81ea4"},"cell_type":"code","source":"print(\"Starting Prediction by paragram Embeddings\")\npred_paragram_val_y = model.predict([val_X], batch_size=1024, verbose=1)\nbest_thresh = 0.5\nbest_score = 0.0\nfor thresh in np.arange(0.1, 0.501, 0.01):\n    thresh = np.round(thresh, 2)\n    score = metrics.f1_score(val_y, (pred_paragram_val_y>thresh).astype(int))\n    print(\"F1 score at threshold {0} is {1}\".format(thresh, metrics.f1_score(val_y, (pred_paragram_val_y>thresh).astype(int))))\n    if score > best_score:\n        best_thresh = thresh\n        best_score = score","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"960b967bf753e8e3a9ab28c9af56c32f66c0c55e"},"cell_type":"code","source":"print()\nprint(\"Best Tresh for paragram is {0} at score {1}\".format(best_thresh,best_score))\nprint()\npred_paragram_val_y = (pred_paragram_val_y>best_thresh)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c1219b4dbf386e8030d091111e6a5d3e9749b05f"},"cell_type":"code","source":"print(\"Prediction for test based on paragram Embeddings\")\npred_paragram_test_y = model.predict([test_X], batch_size=1024, verbose=1)\npred_paragram_test_y = (pred_paragram_test_y>best_thresh)\nprint(\"Paragram Emebddings Prediction Complete\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"7f0759a01a20ec108786917e58a30f55e281cbb7"},"cell_type":"code","source":"del word_index, embeddings_index, all_embs, embedding_matrix, model, inp, x\nimport gc; gc.collect()\ntime.sleep(10)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b1cdc513a1ecb2f7a5c3a4b9c7bc40215ef11ad3"},"cell_type":"code","source":"pred_val_y = 0.34*pred_glove_val_y + 0.33*pred_fasttext_val_y + 0.33*pred_paragram_val_y \n\nbest_thresh = 0.5\nbest_score = 0.0\nfor thresh in np.arange(0.1, 0.501, 0.01):\n    thresh = np.round(thresh, 2)\n    score = metrics.f1_score(val_y, (pred_val_y>thresh).astype(int))\n    print(\"F1 score at threshold {0} is {1}\".format(thresh, metrics.f1_score(val_y, (pred_val_y>thresh).astype(int))))\n    if score > best_score:\n        best_thresh = thresh\n        best_score = score","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"164566a191397afd5ee2997cea97b4bdeb0a4371"},"cell_type":"code","source":"print()\nprint(\"Best Tresh for overall is {0} at score {1}\".format(best_thresh,best_score))\nprint()\npred_val_y = (pred_val_y>best_thresh)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"5471a230caadb709316199adf3564bc690015a59"},"cell_type":"code","source":"pred_test_y = 0.34*pred_glove_test_y + 0.33*pred_fasttext_test_y + 0.33*pred_paragram_test_y\npred_test_y = (pred_test_y>best_thresh).astype(int)\nout_df = pd.DataFrame({\"qid\":test_df[\"qid\"].values})\nout_df['prediction'] = pred_test_y\nout_df.to_csv(\"submission.csv\", index=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f7eaa5519af3eba5ce669c9054927513065955ef"},"cell_type":"code","source":"print(\"Submission Created\")","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}