{"cells":[{"metadata":{},"cell_type":"markdown","source":"# Importing required Libraries"},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"from keras.preprocessing.text import Tokenizer\nfrom keras.preprocessing.sequence import pad_sequences\n\nimport os\nimport pandas as pd\nimport numpy as np\nfrom tqdm import tqdm\nimport math\nfrom sklearn.model_selection import train_test_split","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Importing Dataset"},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df = pd.read_csv(\"../input/train.csv\")\ntrain_df, val_df = train_test_split(train_df,test_size = 0.07)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Let's see the length of the question"},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df.question_text.str.split().str.len().describe()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"From above we can see 75% of question are less than 15 words so let's truncate  the sequence of words"},{"metadata":{"trusted":true},"cell_type":"code","source":"SEQ_LEN = 100 # we set max length of each to be 100 words","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Using glove embeddings"},{"metadata":{"trusted":true},"cell_type":"code","source":"embeddings_index = {}\nf = open('../input/embeddings/glove.840B.300d/glove.840B.300d.txt')\n\nfor line in tqdm(f):\n    values = line.split(\" \")\n    word = values[0]\n    coefs = np.asarray(values[1:],dtype = 'float32')\n    embeddings_index[word] = coefs\nf.close()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Let's see a word vector for the word speech and it's length"},{"metadata":{"trusted":true},"cell_type":"code","source":"print(\"Lenth of vector is \",len(embeddings_index['speech']),\"\\n\",\"Vector for word speech\",\"\\n\",embeddings_index['speech'])","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Let's see the number of word vectors found in glove embeddings"},{"metadata":{"trusted":true},"cell_type":"code","source":"len(embeddings_index)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Data Preprocessing"},{"metadata":{},"cell_type":"markdown","source":"Tokenizing the sentence first"},{"metadata":{"trusted":true},"cell_type":"code","source":"import re\n_WORD_SPLIT = re.compile(\"([.,!?\\\"':;)(])\")\n_DIGIT_RE = re.compile(br\"\\d\")\nSTOP_WORDS = \"\\\" \\' [ ] . , ! : ; ?\".split(\" \")\ndef basic_tokenizer(sentence):\n    \"\"\"Very basic tokenizer: split the sentence into a list of tokens.\"\"\"\n    words = []\n    for space_separated_fragment in sentence.strip().split():\n        words.extend(_WORD_SPLIT.split(space_separated_fragment))\n        # return [w.lower() for w in words if w not in stop_words and w != '' and w != ' ']\n    return [w.lower() for w in words if w != '' and w != ' ']","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Converting the tokenized sentence to embeddings"},{"metadata":{"trusted":true},"cell_type":"code","source":"def text_to_array(text):\n    empyt_emb = np.zeros(300)\n    text = basic_tokenizer(text[:-1])[:SEQ_LEN]\n    embeds = [embeddings_index.get(x, empyt_emb) for x in text]\n    embeds+= [empyt_emb] * (SEQ_LEN - len(embeds))\n    return np.array(embeds)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Applying the preprocessing functions to train and test data"},{"metadata":{"trusted":true},"cell_type":"code","source":"val_vects = np.array([text_to_array(X_text) for X_text in tqdm(val_df[\"question_text\"][:3000])])\nval_y = np.array(val_df[\"target\"][:3000])","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Batching the train data to a size of 256"},{"metadata":{"trusted":true},"cell_type":"code","source":"batch_size = 256\n\ndef batch_gen(train_df):\n    n_batches = math.ceil(len(train_df) / batch_size)\n    while True: \n        train_df = train_df.sample(frac=1.)  # Shuffle the data.\n        for i in range(n_batches):\n            texts = train_df.iloc[i*batch_size:(i+1)*batch_size, 1]\n            text_arr = np.array([text_to_array(text) for text in texts])\n            yield text_arr, np.array(train_df[\"target\"][i*batch_size:(i+1)*batch_size])","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Importing required keras models and layers**"},{"metadata":{"trusted":true},"cell_type":"code","source":"from keras.models import Sequential,Model\nfrom keras.layers import CuDNNLSTM, Dense, Bidirectional,Input,Dropout\n\nfrom keras import backend as K\nfrom keras.engine.topology import Layer\nfrom keras import initializers,regularizers, constraints","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### Creating the Bi Directional LSTM Model"},{"metadata":{"trusted":true},"cell_type":"code","source":"model = Sequential()\nmodel.add(Bidirectional(CuDNNLSTM(128, return_sequences = True),input_shape = (SEQ_LEN,300)))\nmodel.add(Bidirectional(CuDNNLSTM(64)))\nmodel.add(Dense(256,activation = 'relu'))\nmodel.add(Dense(1,activation = 'sigmoid'))\nmodel.compile(loss='binary_crossentropy',\n              optimizer='adam',\n              metrics=['accuracy'])","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Running the model for 5 epochs"},{"metadata":{"trusted":true},"cell_type":"code","source":"mg = batch_gen(train_df)\nmodel.fit_generator(mg, epochs=5,\n                    steps_per_epoch=1000,\n                    validation_data=(val_vects, val_y),\n                    verbose=True)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### Creating the submission file using the test data"},{"metadata":{"trusted":true},"cell_type":"code","source":"batch_size = 256\ndef batch_gen(test_df):\n    n_batches = math.ceil(len(test_df) / batch_size)\n    for i in range(n_batches):\n        texts = test_df.iloc[i*batch_size:(i+1)*batch_size, 1]\n        text_arr = np.array([text_to_array(text) for text in texts])\n        yield text_arr\n\ntest_df = pd.read_csv(\"../input/test.csv\")\n\nall_preds = []\nfor x in tqdm(batch_gen(test_df)):\n    all_preds.extend(model.predict(x).flatten())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"y_te = (np.array(all_preds) > 0.5).astype(np.int)\n\nsubmit_df = pd.DataFrame({\"qid\": test_df[\"qid\"], \"prediction\": y_te})\nsubmit_df.to_csv(\"submission.csv\", index=False)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Saving the model"},{"metadata":{"trusted":true},"cell_type":"code","source":"model_json = model.to_json()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"with open(\"model_questionS.json\", \"w\") as json_file:\n    json_file.write(model_json)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"model.save_weights(\"model_questionS.h5\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.4","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}