#!/usr/bin/env python
# coding: utf-8

import numpy as np
import pandas as pd
import pickle
from collections import Counter
from itertools import chain, product
import gc
import os
import re
from time import time

from sklearn.model_selection import train_test_split
from sklearn.metrics import f1_score

import keras
import keras.backend as K
from keras.callbacks import Callback, ModelCheckpoint, EarlyStopping
from keras.preprocessing.text import Tokenizer
from keras.preprocessing.sequence import pad_sequences
from keras.layers import Dense, Input, Embedding, Dropout, Bidirectional, CuDNNLSTM, CuDNNGRU, SpatialDropout1D, GaussianNoise, Flatten, Conv1D
from keras.models import Model

import gensim
from gensim.models import Word2Vec
from gensim.models.callbacks import CallbackAny2Vec

import spacy
nlp = spacy.load("en_core_web_sm", disable=['tagger', 'parser', 'ner'])

import logging
logging.basicConfig(format='%(asctime)s : %(levelname)s : %(message)s', level=logging.INFO)


t0 = time()

# hyperparameters
embed_size = 64 * 2
max_vocab_size = 95000
maxlen = 64
embedding_epochs = 25
nn_training_epoch = 35
nn_training_batch_size = 1024 * 2
min_count = 2

logging.info('loading data...')
train = pd.read_csv("../input/train.csv")
test = pd.read_csv("../input/test.csv")


logging.info('building vocab...')
questions = pd.concat(
	[train['question_text'], test['question_text']],
	axis=0, sort=False
)
questions = questions.apply(lambda x: x.lower())
questions = questions.str.replace('[0-9]+', '####')
pipeline = nlp.pipe(questions, n_threads=2, batch_size=25000)
corpus = [[token.lemma_ for token in doc] for doc in pipeline]
vocab = Counter((chain(*corpus)))
word_index = {
	token: i + 1
	for i, (token, counts)
	in enumerate(vocab.most_common(max_vocab_size))
	if counts >= min_count
}


logging.info('tokenizing...')
def tokenize(df):
	df = [
		[word_index.get(token, 0) for token in doc if token in word_index]
		for doc in df
	]
	df = pad_sequences(df, maxlen=maxlen, padding="pre", truncating="post")
	return df
	
train_x = corpus[:len(train)]
test_x = corpus[-len(test):]
train_x, test_x = [tokenize(df) for df in [train_x, test_x]]
train_y = train['target'].values


logging.info('prep embedding model...')
class LogScore(CallbackAny2Vec):
	def __init__(self):
		super(LogScore, self)
		self.epoch = 1
		self.last = 0

	def on_epoch_end(self, model):
		loss = model.get_latest_training_loss()
		logging.info(f'Epoch {self.epoch}: {loss - self.last:.4}')
		self.last = loss
		self.epoch += 1
		
embed_model = Word2Vec(
	size=embed_size,
	window=8,
	min_count=min_count,
	sg=1, # use skip-gram
	sample=1e-3,
	negative=3,
	workers=os.cpu_count(),
	batch_words=1024 * 60,
)
embed_model.build_vocab(corpus)
embed_model.train(
	corpus,
	total_examples=len(corpus),
	compute_loss=True,
	epochs=embedding_epochs,
	callbacks=[LogScore()]
)

logging.info('prep embedding embedding weights...')
vocab_size = min(max_vocab_size, len(embed_model.wv.vocab))
embed_matrix = np.zeros((vocab_size + 1, embed_size), dtype=np.float32)
for word, i in word_index.items():
	embed_matrix[i] = embed_model.wv[word]

predictions = 0
for i in range(3):
    train_x, val_x, train_y, val_y = train_test_split(train_x, train_y, test_size=0.2, stratify=train_y)
    
    # build model
    logging.info('build model...')
    token = Input(shape=(maxlen,))
    embedding_layer = Embedding(
    	embed_matrix.shape[0],
    	embed_matrix.shape[1],
    	mask_zero=False,
    	weights=[embed_matrix],
    	input_length=maxlen,
    	trainable=False,
    )
    x = embedding_layer(token)
    dropout_layer = Dropout(.33)
    x = dropout_layer(x)
    x = Bidirectional(CuDNNLSTM(64, return_sequences=True))(x)
    x = Bidirectional(CuDNNGRU(32, return_sequences=True))(x)
    x = Conv1D(1, 1)(x)
    x = Flatten()(x)
    x = Dropout(.2)(x)
    x = Dense(32, activation='elu')(x)
    x = Dense(16, activation='elu')(x)
    x = Dropout(.1)(x)
    out = Dense(1, activation='sigmoid')(x)
    
    model = Model(inputs=token, outputs=out)
    logging.info(model.summary())
    
    class TimeLimits(Callback):
        def __init__(self, time=6400):
            super(TimeLimits, self).__init__()
            self.time = time
        
        def on_epoch_end(self, epoch, logs={}):
        	# save 15 minutes for the model to complete the rest.
        	if time() - t0 > self.time:
        		logging.info("Early stopping due time constraint")
        		self.model.stop_training = True
    
    time_limits_callback = TimeLimits()
    
    checkpoint_callback = ModelCheckpoint(
        './checkpoint.hdf5',
        monitor='val_loss',
        mode = "min",
        save_best_only=True,
    )
    
    def fit(epochs):
    	return model.fit(
    		train_x, train_y,
    		batch_size = nn_training_batch_size,
    		epochs = epochs,
    		validation_data = (val_x, val_y),
    		verbose=2,
    		callbacks = [
    			EarlyStopping(
    			    patience=3,
    			    monitor='val_loss',
    			    restore_best_weights=True
    			),
                checkpoint_callback,
                time_limits_callback,
    		]
	)

    # training stage 1
    adam = keras.optimizers.Adam(lr = 2e-3)
    model.compile(loss='binary_crossentropy', optimizer=adam)
    result = fit(nn_training_epoch)
    model.load_weights('./checkpoint.hdf5')
    
    # training stage 2
    logging.info('unfreezing embedding layer')
    embedding_layer.trainable = True
    dropout_layer.rate = 0.99
    adam = keras.optimizers.Adam(lr = 2e-4)
    model.compile(loss='binary_crossentropy', optimizer=adam)
    result = fit(nn_training_epoch)
    model.load_weights('./checkpoint.hdf5')
                                        

    y_pred = model.predict(val_x)
    thresholds = np.arange(.2, .5, .02)
    scores = [f1_score(val_y, np.where(y_pred >= t, 1, 0)) for t in thresholds]
    best_threshold = thresholds[np.argmax(scores)]
    predictions += np.where(model.predict(test_x, batch_size=nn_training_batch_size) > best_threshold, 1, 0)
    logging.info(best_threshold)
    logging.info(np.max(scores))
    

submission = pd.read_csv('../input/sample_submission.csv')
submission['prediction'] = np.where(predictions / 3 > .5, 1, 0)
submission.to_csv("submission.csv",index=False)



