import os
import time
import numpy as np # linear algebra
import pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)
from tqdm import tqdm
import math
from sklearn.model_selection import train_test_split
from sklearn import metrics

from keras.preprocessing.text import Tokenizer
from keras.preprocessing.sequence import pad_sequences
from keras.layers import Dense, Input, LSTM, Embedding, Dropout, Activation, CuDNNGRU, Conv1D,Flatten,GlobalMaxPooling1D
from keras.layers import Bidirectional, GlobalMaxPool1D
from keras.models import Model
from keras import initializers, regularizers, constraints, optimizers, layers
from keras.models import Sequential


from sklearn.feature_extraction.text import CountVectorizer

train_df1 = pd.read_csv("../input/train.csv")
test_df = pd.read_csv("../input/test.csv")

## split to train and val
train_df, val_df = train_test_split(train_df1, test_size=0.05, random_state=2018)

## some config values 
embed_size = 300 # how big is each word vector
max_features = 70000 # how many unique words to use (i.e num rows in embedding vector)
maxlen = 150 # max number of words in a question to use

puncts = [',', '.', '"', ':', ')', '(', '-', '!', '?', '|', ';', "'", '$', '&', '/', '[', ']', '>', '%', '=', '#', '*', '+', '\\', '•',  '~', '@', '£', 
 '·', '_', '{', '}', '©', '^', '®', '`',  '<', '→', '°', '€', '™', '›',  '♥', '←', '×', '§', '″', '′', 'Â', '█', '½', 'à', '…', 
 '“', '★', '”', '–', '●', 'â', '►', '−', '¢', '²', '¬', '░', '¶', '↑', '±', '¿', '▾', '═', '¦', '║', '―', '¥', '▓', '—', '‹', '─', 
 '▒', '：', '¼', '⊕', '▼', '▪', '†', '■', '’', '▀', '¨', '▄', '♫', '☆', 'é', '¯', '♦', '¤', '▲', 'è', '¸', '¾', 'Ã', '⋅', '‘', '∞', 
 '∙', '）', '↓', '、', '│', '（', '»', '，', '♪', '╩', '╚', '³', '・', '╦', '╣', '╔', '╗', '▬', '❤', 'ï', 'Ø', '¹', '≤', '‡', '√', ]
def clean_text(x):
    x = str(x)
    for punct in puncts:
        x = x.replace(punct, f' {punct} ')
    return x

train_X = train_df1["question_text"].str.lower()
train_X=train_X.apply(lambda x: clean_text(x))
val_X = val_df["question_text"].str.lower()
val_X=val_X.apply(lambda x:clean_text(x))
test_X = test_df["question_text"].str.lower()
test_X=test_X.apply(lambda x: clean_text(x))

## Tokenize the sentences
tokenizer = Tokenizer(num_words=max_features)
tokenizer.fit_on_texts(list(train_X))
train_X = tokenizer.texts_to_sequences(train_X)
val_X = tokenizer.texts_to_sequences(val_X)
test_X = tokenizer.texts_to_sequences(test_X)

# Pad the sentences 
train_X = pad_sequences(train_X, maxlen=maxlen)
val_X = pad_sequences(val_X, maxlen=maxlen)
test_X = pad_sequences(test_X, maxlen=maxlen)

train_y = train_df1['target'].values
val_y = val_df['target'].values

#pre-trained embedding
#glove prediction

EMBEDDING_FILE = '../input/embeddings/glove.840B.300d/glove.840B.300d.txt'
def get_coefs(word,*arr): return word, np.asarray(arr, dtype='float32')
embeddings_index = dict(get_coefs(*o.split(" ")) for o in open(EMBEDDING_FILE))

all_embs = np.stack(embeddings_index.values())
emb_mean,emb_std = all_embs.mean(), all_embs.std()
embed_size = all_embs.shape[1]

word_index = tokenizer.word_index
nb_words = min(max_features, len(word_index))
embedding_matrix = np.random.normal(emb_mean, emb_std, (nb_words, embed_size))
for word, i in word_index.items():
    if i >= max_features: continue
    embedding_vector = embeddings_index.get(word)
    if embedding_vector is not None: embedding_matrix[i] = embedding_vector


model=Sequential([Embedding(max_features,embed_size,input_length=150,weights=[embedding_matrix]),Bidirectional(CuDNNGRU(64, return_sequences=True)),Conv1D(10, kernel_size=5, activation="hard_sigmoid"),GlobalMaxPooling1D(), Dropout(0.1),Dense(256,activation="relu"), Dropout(0.1),Dense(1,activation="sigmoid")])
model.compile(loss='binary_crossentropy', optimizer='adam', metrics=['accuracy'])
#model.compile(loss='sparse_categorical_crossentropy', optimizer='adam',  metrics=['accuracy'])
print(model.summary())


model.fit(train_X, train_y, batch_size=256, epochs=4,   verbose=0,  validation_split=0.03)
score = model.evaluate(val_X,val_y, verbose=1)
print('Test score:', score[0])
print('Test accuracy:', score[1])
predict_glove=model.predict(test_X)


del word_index, embeddings_index, all_embs, embedding_matrix, model
import gc; gc.collect()
time.sleep(10)

#wiki
EMBEDDING_FILE = '../input/embeddings/wiki-news-300d-1M/wiki-news-300d-1M.vec'
def get_coefs(word,*arr): return word, np.asarray(arr, dtype='float32')
embeddings_index = dict(get_coefs(*o.split(" ")) for o in open(EMBEDDING_FILE) if len(o)>100)

all_embs = np.stack(embeddings_index.values())
emb_mean,emb_std = all_embs.mean(), all_embs.std()
embed_size = all_embs.shape[1]

word_index = tokenizer.word_index
nb_words = min(max_features, len(word_index))
embedding_matrix = np.random.normal(emb_mean, emb_std, (nb_words, embed_size))
for word, i in word_index.items():
    if i >= max_features: continue
    embedding_vector = embeddings_index.get(word)
    if embedding_vector is not None: embedding_matrix[i] = embedding_vector

model=Sequential([Embedding(max_features,embed_size,input_length=150,weights=[embedding_matrix]),Bidirectional(CuDNNGRU(64, return_sequences=True)),Conv1D(10, kernel_size=5, activation="hard_sigmoid"),GlobalMaxPooling1D(),Dense(256,activation="relu"), Dropout(0.1),Dense(1,activation="sigmoid")])
model.compile(loss='binary_crossentropy', optimizer='adam', metrics=['accuracy'])
#model.compile(loss='sparse_categorical_crossentropy', optimizer='adam',  metrics=['accuracy'])
print(model.summary())


model.fit(train_X, train_y, batch_size=512, epochs=2,   verbose=0,  validation_split=0.03)
score = model.evaluate(val_X,val_y, verbose=1)
print('Test score:', score[0])
print('Test accuracy:', score[1])
predict_wiki=model.predict(test_X)

del word_index, embeddings_index, all_embs, embedding_matrix, model
import gc; gc.collect()
time.sleep(10)


#paragram
EMBEDDING_FILE = '../input/embeddings/paragram_300_sl999/paragram_300_sl999.txt'
def get_coefs(word,*arr): return word, np.asarray(arr, dtype='float32')
embeddings_index = dict(get_coefs(*o.split(" ")) for o in open(EMBEDDING_FILE, encoding="utf8", errors='ignore') if len(o)>100)

all_embs = np.stack(embeddings_index.values())
emb_mean,emb_std = all_embs.mean(), all_embs.std()
embed_size = all_embs.shape[1]

word_index = tokenizer.word_index
nb_words = min(max_features, len(word_index))
embedding_matrix = np.random.normal(emb_mean, emb_std, (nb_words, embed_size))
for word, i in word_index.items():
    if i >= max_features: continue
    embedding_vector = embeddings_index.get(word)
    if embedding_vector is not None: embedding_matrix[i] = embedding_vector


model=Sequential([Embedding(max_features,embed_size,input_length=150,weights=[embedding_matrix]),Bidirectional(CuDNNGRU(64, return_sequences=True)),Conv1D(10, kernel_size=5, activation="hard_sigmoid"),GlobalMaxPooling1D(),Dense(256,activation="relu"), Dropout(0.1),Dense(1,activation="sigmoid")])
model.compile(loss='binary_crossentropy', optimizer='adam', metrics=['accuracy'])
#model.compile(loss='sparse_categorical_crossentropy', optimizer='adam',  metrics=['accuracy'])
print(model.summary())


model.fit(train_X, train_y, batch_size=512, epochs=2,   verbose=0,  validation_split=0.03)
score = model.evaluate(val_X,val_y, verbose=1)
print('Test score:', score[0])
print('Test accuracy:', score[1])
predict_paragram=model.predict(test_X)

pred_test_y = 0.33*predict_glove + 0.33*predict_wiki + 0.34*predict_paragram
pred1 = (pred_test_y>0.30).astype(int)

result=pd.DataFrame({"qid":test_df["qid"]})
result['prediction'] = pred1
result.to_csv("submission.csv",index=False)
result.to_csv("submission1.csv",index=False)


# inp = Input(shape=(maxlen,))
# x = Embedding(max_features, embed_size,weights=[embedding_matrix])(inp)
# x = Bidirectional(CuDNNGRU(64, return_sequences=True))(x)
# x = Conv1D(10, kernel_size=5, activation="hard_sigmoid")(x)
# x = Conv1D(4, kernel_size=2, activation="sigmoid")(x)
# x = GlobalMaxPool1D()(x)
# x = Dense(16, activation="relu")(x)
# x = Dropout(0.1)(x)
# x = Dense(1, activation="sigmoid")(x)
# model = Model(inputs=inp, outputs=x)
# model.compile(loss='binary_crossentropy', optimizer='adam', metrics=['accuracy'])

# print(model.summary())
