{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nimport os\nprint(os.listdir(\"../input\"))","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"3153d1417bb7cf1f363d0f69c110c3a0aa4ece1a"},"cell_type":"markdown","source":"### Define embedding size and maximum features you need for your model"},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"## some config values \nembed_size = 300 # how big is each word vector\nmax_features = 95000 # how many unique words to use (i.e num rows in embedding vector)\nmaxlen = 70 # max number of words in a question to use","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"5218a5b6340a7cb3337c0298f47b44839b1ca4f8"},"cell_type":"markdown","source":"### Load additional libraries"},{"metadata":{"trusted":true,"_uuid":"798b7c8319c2b205b20b25e062f3a8582b4b1f29"},"cell_type":"code","source":"import os\nimport time\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nfrom tqdm import tqdm\nimport math\nfrom sklearn.model_selection import train_test_split\nfrom sklearn import metrics\nimport string, re\nfrom keras.preprocessing.text import Tokenizer\nfrom keras.preprocessing.sequence import pad_sequences\nfrom keras.layers import Dense, Input, CuDNNLSTM, Embedding, Dropout, Activation, CuDNNGRU, Conv1D\nfrom keras.layers import Bidirectional, GlobalMaxPool1D, GlobalMaxPooling1D, GlobalAveragePooling1D\nfrom keras.layers import Input, Embedding, Dense, Conv2D, MaxPool2D, concatenate\nfrom keras.layers import Reshape, Flatten, Concatenate, Dropout, SpatialDropout1D\nfrom keras.optimizers import Adam\nfrom keras.models import Model\nfrom keras import backend as K\nfrom keras.engine.topology import Layer\nfrom keras import initializers, regularizers, constraints, optimizers, layers","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"669beb0792a65e86c9c622f7e0a2a8489fca4ad6"},"cell_type":"markdown","source":"### Do some cleaning to make your model more effective"},{"metadata":{"trusted":true,"_uuid":"d47bacc873b2de2c56ae39f41beaceac01527823"},"cell_type":"code","source":"def clean(text):\n    \n    ## Remove puncuation\n    text = text.translate(string.punctuation)\n    \n    ## Convert words to lower case and split them\n    text = text.lower()\n    \n    ## Remove stop words\n    #text = text.split()\n    #stops = set(stopwords.words(\"english\"))\n    #text = [w for w in text if not w in stops and len(w) >= 3]\n    \n    #text = \" \".join(text)\n\n    # Clean the text\n    text = re.sub(r\"[^A-Za-z0-9^,!.\\/'+-=]\", \" \", text)\n    text = re.sub(r\"what's\", \"what is \", text)\n    text = re.sub(r\"\\'s\", \" \", text)\n    text = re.sub(r\"\\'ve\", \" have \", text)\n    text = re.sub(r\"n't\", \" not \", text)\n    text = re.sub(r\"i'm\", \"i am \", text)\n    text = re.sub(r\"\\'re\", \" are \", text)\n    text = re.sub(r\"\\'d\", \" would \", text)\n    text = re.sub(r\"\\'ll\", \" will \", text)\n    text = re.sub(r\",\", \" \", text)\n    text = re.sub(r\"\\.\", \" \", text)\n    text = re.sub(r\"!\", \" ! \", text)\n    text = re.sub(r\"\\/\", \" \", text)\n    text = re.sub(r\"\\^\", \" ^ \", text)\n    text = re.sub(r\"\\+\", \" + \", text)\n    text = re.sub(r\"\\-\", \" - \", text)\n    text = re.sub(r\"\\=\", \" = \", text)\n    text = re.sub(r\"'\", \" \", text)\n    text = re.sub(r\"(\\d+)(k)\", r\"\\g<1>000\", text)\n    text = re.sub(r\":\", \" : \", text)\n    text = re.sub(r\" e g \", \" eg \", text)\n    text = re.sub(r\" b g \", \" bg \", text)\n    text = re.sub(r\" u s \", \" american \", text)\n    text = re.sub(r\"\\0s\", \"0\", text)\n    text = re.sub(r\" 9 11 \", \"911\", text)\n    text = re.sub(r\"e - mail\", \"email\", text)\n    text = re.sub(r\"j k\", \"jk\", text)\n    text = re.sub(r\"\\s{2,}\", \" \", text)\n    text = re.sub('[^a-zA-Z]',' ', text)\n    text = re.sub('  +',' ',text)\n    \n    #text = text.split()\n    #stemmer = SnowballStemmer('english')\n    #stemmed_words = [stemmer.stem(word) for word in text]\n    #text = \" \".join(stemmed_words)\n    return text","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"c7781f09449515638f559183bf6eb86b254a4496"},"cell_type":"markdown","source":"### Load training and test data : convert into padded sentences for input to keras model"},{"metadata":{"trusted":true,"_uuid":"98628d1c95090a19f36011325d1061c40b680c74"},"cell_type":"code","source":"def load_and_prec():\n    train_df = pd.read_csv(\"../input/train.csv\")\n    test_df = pd.read_csv(\"../input/test.csv\")\n    print(\"Train shape : \",train_df.shape)\n    print(\"Test shape : \",test_df.shape)\n    \n    train_df['clean_text'] = train_df['question_text'].apply(clean)\n    test_df['clean_text'] = test_df['question_text'].apply(clean)\n    \n    ## split to train and val\n    train_df, val_df = train_test_split(train_df, test_size=0.08, random_state=2018)\n\n\n    ## fill up the missing values\n    train_X = train_df[\"clean_text\"].fillna(\"_##_\").values\n    val_X = val_df[\"clean_text\"].fillna(\"_##_\").values\n    test_X = test_df[\"clean_text\"].fillna(\"_##_\").values\n    \n    ## Tokenize the sentences\n    tokenizer = Tokenizer(num_words=max_features)\n    tokenizer.fit_on_texts(list(train_X))\n    train_X = tokenizer.texts_to_sequences(train_X)\n    val_X = tokenizer.texts_to_sequences(val_X)\n    test_X = tokenizer.texts_to_sequences(test_X)\n\n    ## Pad the sentences \n    train_X = pad_sequences(train_X, maxlen=maxlen)\n    val_X = pad_sequences(val_X, maxlen=maxlen)\n    test_X = pad_sequences(test_X, maxlen=maxlen)\n\n    ## Get the target values\n    train_y = train_df['target'].values\n    val_y = val_df['target'].values  \n    \n    #shuffling the data\n    np.random.seed(2018)\n    trn_idx = np.random.permutation(len(train_X))\n    val_idx = np.random.permutation(len(val_X))\n\n    train_X = train_X[trn_idx]\n    val_X = val_X[val_idx]\n    train_y = train_y[trn_idx]\n    val_y = val_y[val_idx]    \n    \n    return train_X, val_X, test_X, train_y, val_y, tokenizer.word_index","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"9542951a112f495625c13f5799c7cab704d692e2"},"cell_type":"markdown","source":"### Define functions to load different embeddings, we will be using only glove for this basic model"},{"metadata":{"trusted":true,"_uuid":"f8da734c7283ee393c47b05dee8b631e8fa3c333"},"cell_type":"code","source":"def load_glove(word_index):\n    EMBEDDING_FILE = '../input/embeddings/glove.840B.300d/glove.840B.300d.txt'\n    def get_coefs(word,*arr): return word, np.asarray(arr, dtype='float32')\n    embeddings_index = dict(get_coefs(*o.split(\" \")) for o in open(EMBEDDING_FILE))\n\n    all_embs = np.stack(embeddings_index.values())\n    emb_mean,emb_std = all_embs.mean(), all_embs.std()\n    embed_size = all_embs.shape[1]\n\n    # word_index = tokenizer.word_index\n    nb_words = min(max_features, len(word_index))\n    embedding_matrix = np.random.normal(emb_mean, emb_std, (nb_words, embed_size))\n    for word, i in word_index.items():\n        if i >= max_features: continue\n        embedding_vector = embeddings_index.get(word)\n        if embedding_vector is not None: embedding_matrix[i] = embedding_vector\n            \n    return embedding_matrix \n    \ndef load_fasttext(word_index):    \n    EMBEDDING_FILE = '../input/embeddings/wiki-news-300d-1M/wiki-news-300d-1M.vec'\n    def get_coefs(word,*arr): return word, np.asarray(arr, dtype='float32')\n    embeddings_index = dict(get_coefs(*o.split(\" \")) for o in open(EMBEDDING_FILE) if len(o)>100)\n\n    all_embs = np.stack(embeddings_index.values())\n    emb_mean,emb_std = all_embs.mean(), all_embs.std()\n    embed_size = all_embs.shape[1]\n\n    # word_index = tokenizer.word_index\n    nb_words = min(max_features, len(word_index))\n    embedding_matrix = np.random.normal(emb_mean, emb_std, (nb_words, embed_size))\n    for word, i in word_index.items():\n        if i >= max_features: continue\n        embedding_vector = embeddings_index.get(word)\n        if embedding_vector is not None: embedding_matrix[i] = embedding_vector\n\n    return embedding_matrix\n\ndef load_para(word_index):\n    EMBEDDING_FILE = '../input/embeddings/paragram_300_sl999/paragram_300_sl999.txt'\n    def get_coefs(word,*arr): return word, np.asarray(arr, dtype='float32')\n    embeddings_index = dict(get_coefs(*o.split(\" \")) for o in open(EMBEDDING_FILE, encoding=\"utf8\", errors='ignore') if len(o)>100)\n\n    all_embs = np.stack(embeddings_index.values())\n    emb_mean,emb_std = all_embs.mean(), all_embs.std()\n    embed_size = all_embs.shape[1]\n\n    # word_index = tokenizer.word_index\n    nb_words = min(max_features, len(word_index))\n    embedding_matrix = np.random.normal(emb_mean, emb_std, (nb_words, embed_size))\n    for word, i in word_index.items():\n        if i >= max_features: continue\n        embedding_vector = embeddings_index.get(word)\n        if embedding_vector is not None: embedding_matrix[i] = embedding_vector\n    \n    return embedding_matrix\n\n","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"9966b5673425de2cf5a7c8772641e3c46ff84679"},"cell_type":"markdown","source":"### Define LSTM model architecture"},{"metadata":{"trusted":true,"_uuid":"90ba931507cf43a11939fe864dcf6dc5dd82bf29"},"cell_type":"code","source":"def model_lstm_du(embedding_matrix):\n    inp = Input(shape=(maxlen,))\n    x = Embedding(max_features, embed_size, weights=[embedding_matrix])(inp)\n    x = Bidirectional(CuDNNGRU(64, return_sequences=True))(x)\n    avg_pool = GlobalAveragePooling1D()(x)\n    max_pool = GlobalMaxPooling1D()(x)\n    conc = concatenate([avg_pool, max_pool])\n    conc = Dense(64, activation=\"relu\")(conc)\n    conc = Dropout(0.1)(conc)\n    outp = Dense(1, activation=\"sigmoid\")(conc)\n    \n    model = Model(inputs=inp, outputs=outp)\n    model.compile(loss='binary_crossentropy', optimizer='adam', metrics=['accuracy'])\n    return model","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"f7b5a6007fdac8d804aeaedfecb12fab01e9b210"},"cell_type":"markdown","source":"### Define training function"},{"metadata":{"trusted":true,"_uuid":"f6b0e66affb5c91a20460761f0465c45d07d6a67"},"cell_type":"code","source":"def train_pred(model, epochs=2):\n    for e in range(epochs):\n        model.fit(train_X, train_y, batch_size=512, epochs=1, validation_data=(val_X, val_y))\n        pred_val_y = model.predict([val_X], batch_size=1024, verbose=0)\n\n        best_thresh = 0.5\n        best_score = 0.0\n        for thresh in np.arange(0.1, 0.501, 0.01):\n            thresh = np.round(thresh, 2)\n            score = metrics.f1_score(val_y, (pred_val_y > thresh).astype(int))\n            if score > best_score:\n                best_thresh = thresh\n                best_score = score\n\n        print(\"Val F1 Score: {:.4f}\".format(best_score))\n\n    pred_test_y = model.predict([test_X], batch_size=1024, verbose=0)\n    return pred_val_y, pred_test_y, best_score","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"f6eb18a4466596948ed87a7f4657e7a13d71d5b0"},"cell_type":"markdown","source":"### Load the data"},{"metadata":{"trusted":true,"_uuid":"427ec406855216d63edbe6e99b9c8be3c5aa611b"},"cell_type":"code","source":"%%time\ntrain_X, val_X, test_X, train_y, val_y, word_index = load_and_prec()\n","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"202e457bc3b1c31d6c14d0b93110950104c7f555"},"cell_type":"markdown","source":"### load the embeddings"},{"metadata":{"trusted":true,"_uuid":"60d059aafb688cd8243c2d841097974d49b1fedc"},"cell_type":"code","source":"%%time\nembedding_matrix_1 = load_glove(word_index)\n# embedding_matrix_2 = load_fasttext(word_index)\n#embedding_matrix_3 = load_para(word_index)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"56817af9a183fa56afe409d64841f7db634fecd0"},"cell_type":"markdown","source":"### Run the model"},{"metadata":{"trusted":true,"_uuid":"5d077b8406f24c548085dbf5c6074ce4ee405d28"},"cell_type":"code","source":"%%time\npred_val_y, pred_test_y, best_score = train_pred(model_lstm_du(embedding_matrix_1), epochs = 2)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"2eaa2f9abd52e9cd7bcc8368411917b82e8965f9"},"cell_type":"markdown","source":"### Choose best threshold"},{"metadata":{"trusted":true,"_uuid":"729abe3d6329907544364a55009ad74b996fd659"},"cell_type":"code","source":"thresholds = []\nfor thresh in np.arange(0.1, 0.501, 0.01):\n    thresh = np.round(thresh, 2)\n    res = metrics.f1_score(val_y, (pred_val_y > thresh).astype(int))\n    thresholds.append([thresh, res])\n    print(\"F1 score at threshold {0} is {1}\".format(thresh, res))\n    \nthresholds.sort(key=lambda x: x[1], reverse=True)\nbest_thresh = thresholds[0][0]\nprint(\"Best threshold: \", best_thresh)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"bfc5c1ad92808b3292c1e0e0aa5f3963e769c036"},"cell_type":"markdown","source":"### Confusion matrix of validation output"},{"metadata":{"trusted":true,"_uuid":"f7260986ad8781817104d8287eae3cd4857ab2c5"},"cell_type":"code","source":"metrics.confusion_matrix(val_y,pred_val_y>best_thresh)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"9a20145493992659ae06dd20eab91ef93a8c3a2d"},"cell_type":"markdown","source":"### Predict on test and save the output (using the besst threshold chosen above)"},{"metadata":{"trusted":true,"_uuid":"a5d72ce28f0f4b338c9e76a37ae9a2a0583a6679"},"cell_type":"code","source":"pred_test_y = (pred_test_y > best_thresh).astype(int)\ntest_df = pd.read_csv(\"../input/test.csv\", usecols=[\"qid\"])\nout_df = pd.DataFrame({\"qid\":test_df[\"qid\"].values})\nout_df['prediction'] = pred_test_y\nout_df.to_csv(\"submission.csv\", index=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a090f4f6770013659964d80e9a137752016cde20"},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d2dd51260cdc6b3f1c350ee6cdee973cee3bbd2d"},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}