import os
import time
import numpy as np # linear algebra
import pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)
from tqdm import tqdm
import math
import operator 

from sklearn.model_selection import train_test_split
from sklearn import metrics

from keras.preprocessing.text import Tokenizer
from keras.preprocessing.sequence import pad_sequences
from keras.layers import Dense, Input, LSTM, Embedding, Dropout, Activation, CuDNNGRU, Conv1D
from keras.layers import Bidirectional, GlobalMaxPool1D,GlobalMaxPooling1D
from keras.models import Model,Sequential
from keras import initializers, regularizers, constraints, optimizers, layers
from gensim.models import KeyedVectors

## some config values 
embed_size = 300 # how big is each word vector
max_features = 55000 # how many unique words to use (i.e num rows in embedding vector)
maxlen = 150 # max number of words in a question to use

import re

def clean_numbers(x):

    x = re.sub('[0-9]{5,}', '#####', x)
    x = re.sub('[0-9]{4}', '####', x)
    x = re.sub('[0-9]{3}', '###', x)
    x = re.sub('[0-9]{2}', '##', x)
    return x

def clean_text(x):

    x = str(x)
    for punct in "/-'":
        x = x.replace(punct, ' ')
    for punct in '&':
        x = x.replace(punct, f' {punct} ')
    for punct in '?!.,"#$%\'()*+-/:;<=>@[\\]^_`{|}~' + '“”’':
        x = x.replace(punct, '')
    return x

def _get_mispell(mispell_dict):
    mispell_re = re.compile('(%s)' % '|'.join(mispell_dict.keys()))
    return mispell_dict, mispell_re


mispell_dict = {'colour':'color',
                'centre':'center',
                'didnt':'did not',
                'doesnt':'does not',
                'isnt':'is not',
                'shouldnt':'should not',
                'favourite':'favorite',
                'travelling':'traveling',
                'counselling':'counseling',
                'theatre':'theater',
                'cancelled':'canceled',
                'labour':'labor',
                'organisation':'organization',
                'wwii':'world war 2',
                'citicise':'criticize',
                'instagram': 'social medium',
                'whatsapp': 'social medium',
                'snapchat': 'social medium'

                }
mispellings, mispellings_re = _get_mispell(mispell_dict)

def replace_typical_misspell(text):
    def replace(match):
        return mispellings[match.group(0)]

    return mispellings_re.sub(replace, text)

def build_vocab(sentences, verbose =  True):
    """
    :param sentences: list of list of words
    :return: dictionary of words and their count
    """
    vocab = {}
    for sentence in tqdm(sentences, disable = (not verbose)):
        for word in sentence:
            try:
                vocab[word] += 1
            except KeyError:
                vocab[word] = 1
    return vocab
def remove_word(x):
    x = re.sub(' a ', ' ', x)
    x = re.sub(' to ', ' ', x)
    x = re.sub(' and ', ' ', x)
    x = re.sub(' of ',' ',x)
    return x
    

train = pd.read_csv("../input/train.csv")
test = pd.read_csv("../input/test.csv")
tqdm.pandas()
train_y = train['target'].values
train_qt=train["question_text"].apply(lambda x: clean_text(x))
train_qt=train_qt.apply(lambda x: clean_numbers(x))
train_qt=train_qt.apply(lambda x: replace_typical_misspell(x))
train_qt=train_qt.apply(lambda x: remove_word(x))


test_qt=test["question_text"].apply(lambda x: clean_text(x))
test_qt=test_qt.apply(lambda x: clean_numbers(x))
test_qt=test_qt.apply(lambda x: replace_typical_misspell(x))
test_qt=test_qt.apply(lambda x: remove_word(x))
sentences = test_qt.apply(lambda x: x.split()).values
# vocab = build_vocab(sentences)
# print({k: vocab[k] for k in list(vocab)[:5]})

#clear the memory   
# del sentences,train,test
# import gc
# gc.collect()

## some config values 
embed_size = 300 # how big is each word vector
max_features = 60000 # how many unique words to use (i.e num rows in embedding vector)
maxlen = 150 # max number of words in a question to use

## Tokenize the sentences
tokenizer = Tokenizer(num_words=max_features)
tokenizer.fit_on_texts(list(train_qt))
train_X = tokenizer.texts_to_sequences(train_qt)
test_X = tokenizer.texts_to_sequences(test_qt)

# Pad the sentences 
train_X = pad_sequences(train_X, maxlen=maxlen)
test_X = pad_sequences(test_X, maxlen=maxlen)

# del train_qt,test_qt
# news_path = '../input/embeddings/GoogleNews-vectors-negative300/GoogleNews-vectors-negative300.bin'
# embeddings_index = KeyedVectors.load_word2vec_format(news_path, binary=True)


# all_embs = np.stack(embeddings_index.vectors)
# emb_mean,emb_std = all_embs.mean(), all_embs.std()
# embed_size = all_embs.shape[1]

# word_index = tokenizer.word_index
# nb_words = min(max_features, len(word_index))
# embedding_matrix = np.random.normal(emb_mean, emb_std, (nb_words, embed_size))
# not_in_embeddings=[]
# for word, i in word_index.items():
#     if i >= max_features: continue
#     try:
#         embedding_vector = embeddings_index[word]
#         if embedding_vector is not None: embedding_matrix[i] = embedding_vector
#     except:
#         not_in_embeddings.append(word)




EMBEDDING_FILE = '../input/embeddings/glove.840B.300d/glove.840B.300d.txt'
def get_coefs(word,*arr): return word, np.asarray(arr, dtype='float32')
embeddings_index = dict(get_coefs(*o.split(" ")) for o in open(EMBEDDING_FILE))

all_embs = np.stack(embeddings_index.values())
emb_mean,emb_std = all_embs.mean(), all_embs.std()
embed_size = all_embs.shape[1]

word_index = tokenizer.word_index
nb_words = min(max_features, len(word_index))
embedding_matrix = np.random.normal(emb_mean, emb_std, (nb_words, embed_size))
for word, i in word_index.items():
    if i >= max_features: continue
    embedding_vector = embeddings_index.get(word)
    if embedding_vector is not None: embedding_matrix[i] = embedding_vector
    
del embeddings_index,train_qt,test_qt,sentences
import gc
gc.collect()
time.sleep(10)

model=Sequential([Embedding(max_features,embed_size,input_length=150,weights=[embedding_matrix]),Bidirectional(CuDNNGRU(64, return_sequences=True)),Conv1D(10, kernel_size=5, activation="hard_sigmoid"),GlobalMaxPooling1D(),Dense(256,activation="relu"), Dropout(0.1),Dense(1,activation="sigmoid")])
model.compile(loss='binary_crossentropy', optimizer='adam', metrics=['accuracy'])
#model.compile(loss='sparse_categorical_crossentropy', optimizer='adam',  metrics=['accuracy'])
print(model.summary())

model.fit(train_X, train_y, batch_size=512, epochs=2,   verbose=0,  validation_split=0.03)
predict_google=model.predict(test_X)

del model
import gc
gc.collect()
time.sleep(10)


EMBEDDING_FILE = '../input/embeddings/glove.840B.300d/glove.840B.300d.txt'
def get_coefs(word,*arr): return word, np.asarray(arr, dtype='float32')
embeddings_index = dict(get_coefs(*o.split(" ")) for o in open(EMBEDDING_FILE))

all_embs = np.stack(embeddings_index.values())
emb_mean,emb_std = all_embs.mean(), all_embs.std()
embed_size = all_embs.shape[1]

word_index = tokenizer.word_index
nb_words = min(max_features, len(word_index))
embedding_matrix = np.random.normal(emb_mean, emb_std, (nb_words, embed_size))
for word, i in word_index.items():
    if i >= max_features: continue
    embedding_vector = embeddings_index.get(word)
    if embedding_vector is not None: embedding_matrix[i] = embedding_vector


model=Sequential([Embedding(max_features,embed_size,input_length=150,weights=[embedding_matrix]),Bidirectional(CuDNNGRU(64, return_sequences=True)),Conv1D(10, kernel_size=5, activation="hard_sigmoid"),GlobalMaxPooling1D(), Dropout(0.1),Dense(256,activation="relu"), Dropout(0.1),Dense(1,activation="sigmoid")])
model.compile(loss='binary_crossentropy', optimizer='adam', metrics=['accuracy'])
#model.compile(loss='sparse_categorical_crossentropy', optimizer='adam',  metrics=['accuracy'])
print(model.summary())


model.fit(train_X, train_y, batch_size=256, epochs=4,   verbose=0,  validation_split=0.03)
predict_glove=model.predict(test_X)


del word_index, embeddings_index, all_embs, embedding_matrix, model
import gc; gc.collect()
time.sleep(10)




#wiki
EMBEDDING_FILE = '../input/embeddings/wiki-news-300d-1M/wiki-news-300d-1M.vec'
def get_coefs(word,*arr): return word, np.asarray(arr, dtype='float32')
embeddings_index = dict(get_coefs(*o.split(" ")) for o in open(EMBEDDING_FILE) if len(o)>100)

all_embs = np.stack(embeddings_index.values())
emb_mean,emb_std = all_embs.mean(), all_embs.std()
embed_size = all_embs.shape[1]

word_index = tokenizer.word_index
nb_words = min(max_features, len(word_index))
embedding_matrix = np.random.normal(emb_mean, emb_std, (nb_words, embed_size))
for word, i in word_index.items():
    if i >= max_features: continue
    embedding_vector = embeddings_index.get(word)
    if embedding_vector is not None: embedding_matrix[i] = embedding_vector

model=Sequential([Embedding(max_features,embed_size,input_length=150,weights=[embedding_matrix]),Bidirectional(CuDNNGRU(64, return_sequences=True)),Conv1D(10, kernel_size=5, activation="hard_sigmoid"),GlobalMaxPooling1D(),Dense(256,activation="relu"), Dropout(0.1),Dense(1,activation="sigmoid")])
model.compile(loss='binary_crossentropy', optimizer='adam', metrics=['accuracy'])
#model.compile(loss='sparse_categorical_crossentropy', optimizer='adam',  metrics=['accuracy'])
print(model.summary())


model.fit(train_X, train_y, batch_size=512, epochs=2,   verbose=0,  validation_split=0.03)
predict_wiki=model.predict(test_X)

del word_index, embeddings_index, all_embs, embedding_matrix, model
import gc; gc.collect()
time.sleep(10)


#paragram
EMBEDDING_FILE = '../input/embeddings/paragram_300_sl999/paragram_300_sl999.txt'
def get_coefs(word,*arr): return word, np.asarray(arr, dtype='float32')
embeddings_index = dict(get_coefs(*o.split(" ")) for o in open(EMBEDDING_FILE, encoding="utf8", errors='ignore') if len(o)>100)

all_embs = np.stack(embeddings_index.values())
emb_mean,emb_std = all_embs.mean(), all_embs.std()
embed_size = all_embs.shape[1]

word_index = tokenizer.word_index
nb_words = min(max_features, len(word_index))
embedding_matrix = np.random.normal(emb_mean, emb_std, (nb_words, embed_size))
for word, i in word_index.items():
    if i >= max_features: continue
    embedding_vector = embeddings_index.get(word)
    if embedding_vector is not None: embedding_matrix[i] = embedding_vector


model=Sequential([Embedding(max_features,embed_size,input_length=150,weights=[embedding_matrix]),Bidirectional(CuDNNGRU(64, return_sequences=True)),Conv1D(10, kernel_size=5, activation="hard_sigmoid"),GlobalMaxPooling1D(),Dense(256,activation="relu"), Dropout(0.1),Dense(1,activation="sigmoid")])
model.compile(loss='binary_crossentropy', optimizer='adam', metrics=['accuracy'])
#model.compile(loss='sparse_categorical_crossentropy', optimizer='adam',  metrics=['accuracy'])
print(model.summary())


model.fit(train_X, train_y, batch_size=512, epochs=2,   verbose=0,  validation_split=0.03)
predict_paragram=model.predict(test_X)

test = pd.read_csv("../input/test.csv")

pred_test_y = 0.35*predict_glove + 0.40*predict_google + 0.15*predict_wiki + 0.10*predict_paragram
predict = (pred_test_y>0.25).astype(int)
result=pd.DataFrame({"qid":test["qid"]})
result['prediction'] = predict
result.to_csv("submission.csv",index=False) 