# This Python 3 environment comes with many helpful analytics libraries installed
# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python
# For example, here's several helpful packages to load in 

import numpy as np # linear algebra
import pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)
# Input data files are available in the "../input/" directory.
# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory
import os
#print(os.listdir("../input"))
# Any results you write to the current directory are saved as output.
from subprocess import check_output
print(check_output(["ls", "../input"]).decode("utf8"))

import csv
import re
import tensorflow as tf
from collections import defaultdict,OrderedDict
import gensim
import random

# imports for preprocessing the questions
from keras.preprocessing.text import Tokenizer
from keras.preprocessing.sequence import pad_sequences
from sklearn.model_selection import train_test_split

# cross validation and metrics
from sklearn.model_selection import StratifiedKFold
from sklearn.metrics import f1_score

# progress bars
from tqdm import tqdm
tqdm.pandas()

class Data_pro():

    def __init__(self):
        self.puncts = [',', '.', '"', ':', ')', '(', '-', '!', '?', '|', ';', "'", '$', '&', '/', '[', ']', '>', '%', '=', '#', '*', '+', '\\', '•',  '~', '@', '£', 
    '·', '_', '{', '}', '©', '^', '®', '`',  '<', '→', '°', '€', '™', '›',  '♥', '←', '×', '§', '″', '′', 'Â', '█', '½', 'à', '…', 
    '“', '★', '”', '–', '●', 'â', '►', '−', '¢', '²', '¬', '░', '¶', '↑', '±', '¿', '▾', '═', '¦', '║', '―', '¥', '▓', '—', '‹', '─', 
    '▒', '：', '¼', '⊕', '▼', '▪', '†', '■', '’', '▀', '¨', '▄', '♫', '☆', 'é', '¯', '♦', '¤', '▲', 'è', '¸', '¾', 'Ã', '⋅', '‘', '∞', 
    '∙', '）', '↓', '、', '│', '（', '»', '，', '♪', '╩', '╚', '³', '・', '╦', '╣', '╔', '╗', '▬', '❤', 'ï', 'Ø', '¹', '≤', '‡', '√', ]
        self.maxlen = 72
        self.max_words = 100000
        self.x_train, self.y_train, self.x_val, self.y_val, self.x_test, self.qid_test = self.get_csv_inf()
        self.token = self.get_vocab(self.x_train)
        self.x_train_ids, self.x_val_ids, self.x_test_ids = self.get_word_ids(self.token, self.x_train, self.x_val, self.x_test)
        #self.words_vec = self.embedding_func()
        self.wlist = self.embedding_func_glove()
        

    def clean_text(self,x):
        x = str(x)
        for punct in self.puncts:
            x = x.replace(punct, f' {punct} ')
        return x

    def get_csv_inf(self):

        PATH = "../input/"
        train_df = pd.read_csv(PATH + "train.csv",engine='python',nrows=None,encoding='utf-8')
        test_df = pd.read_csv(PATH + "test.csv",engine='python',nrows=None,encoding='utf-8')

        train_df["question_text"] = train_df["question_text"].str.lower()
        test_df["question_text"] = test_df["question_text"].str.lower()

        train_df["question_text"] = train_df["question_text"].apply(lambda x: self.clean_text(x))
        test_df["question_text"]  = test_df["question_text"].apply(lambda x: self.clean_text(x))

        train_df, val_df = train_test_split(train_df, test_size=0.2, random_state=2018)

        # fill up the missing values
        x_train = train_df["question_text"].fillna("_##_").values
        x_val = val_df["question_text"].fillna("_##_").values

        x_test = test_df["question_text"].fillna("_##_").values
        qid = test_df["qid"].values

        y_train = train_df['target'].values
        y_val = val_df['target'].values

        #train_y1 = np.array(train_y, dtype="int")
        train_y2 = list(map(self.get_contrast, y_train))
        train_y = np.array([y_train, train_y2]).T

        #test_y1 = np.array(test_y, dtype="int")
        val_y2 = list(map(self.get_contrast, y_val))
        val_y = np.array([y_val, val_y2]).T
      
        return x_train, train_y, x_val, val_y, x_test, qid
 
   
    def get_contrast(self,x):
        return (2+~x)

    def get_vocab(self, x_train):
        tokenizer = Tokenizer(num_words=self.max_words)
        tokenizer.fit_on_texts(list(x_train))

        return tokenizer

    def get_word_ids(self, token, x_train, x_val, x_test):
        x_train_ids = token.texts_to_sequences(x_train)
        x_val_ids = token.texts_to_sequences(x_val)
        x_test_ids = token.texts_to_sequences(x_test)

        # Pad the sentences 
        x_train_ids = pad_sequences(x_train_ids, maxlen=self.maxlen,padding='post')
        x_val_ids = pad_sequences(x_val_ids, maxlen=self.maxlen,padding='post')
        x_test_ids = pad_sequences(x_test_ids, maxlen=self.maxlen,padding='post')

        return x_train_ids, x_val_ids, x_test_ids

    def embedding_func(self):

        model = gensim.models.KeyedVectors.load_word2vec_format("../input/" + 'embeddings/GoogleNews-vectors-negative300/GoogleNews-vectors-negative300.bin',binary=True)
        words_vec = {} 
        vocab = self.token.word_index

        for word,index in vocab.items():
            if word in model.vocab:
                words_vec[word] = model[word]
            else:
                words_vec[word] = np.random.random((300,))
                #words_vec[word] = np.zeros((300,))
        
        #words_vec["UNK"] = np.random.random((300,))

        return words_vec
    
    def embedding_func_glove(self):

        embedding_mat = self.load_glove(self.token.word_index)
        # words_vec = {} 

        # for word,index in vocab.items():
        #     if word in model.vocab:
        #         words_vec[word] = model[word]
        #     else:
        #         words_vec[word] = np.random.random((300,))
              

        # return words_vec
        return embedding_mat
    
    def get_words_vec(self):
        W_list = []
        W_list.append([0.0]*300)

        for word,vector in self.words_vec.items():
            W_list.append(vector.tolist())
        
        return W_list
    
    def load_glove(self, word_index):
      
        EMBEDDING_FILE = '../input/embeddings/glove.840B.300d/glove.840B.300d.txt'
        def get_coefs(word,*arr): return word, np.asarray(arr, dtype='float32')
        embeddings_index = dict(get_coefs(*o.split(" ")) for o in open(EMBEDDING_FILE,encoding='utf-8'))

        all_embs = np.stack(embeddings_index.values())
        emb_mean,emb_std = -0.005838499,0.48782197
        embed_size = all_embs.shape[1]

        # word_index = tokenizer.word_index
        nb_words = min(self.max_words, len(word_index))
        embedding_matrix = np.random.normal(emb_mean, emb_std, (nb_words, embed_size))
        for word, i in word_index.items():
            if i >= self.max_words: continue
            embedding_vector = embeddings_index.get(word)
            if embedding_vector is not None: embedding_matrix[i] = embedding_vector

        embedding_matrix[0] = np.zeros((300,))
        embedding_matrix = embedding_matrix.tolist()
        return embedding_matrix 


class TextCNN:
    def __init__(self,W_list,embd_train_bool,sequence_length,num_class,embedding_size,filter_sizes,num_fliters,batch_size,l2_reg):
        self.input_x = tf.placeholder(tf.int32,[None,sequence_length])
        self.input_y = tf.placeholder(tf.int32,[None,num_class])
        self.drop_prob = tf.placeholder(tf.float32)

        l2_loss = tf.constant(0.0)
        with tf.name_scope("embedding"):
            W = tf.Variable(initial_value = W_list, dtype = tf.float32, trainable = embd_train_bool)
            self.embedding_chars = tf.nn.embedding_lookup(W,self.input_x)
            self.embedding_chars_exp = tf.expand_dims(self.embedding_chars,-1)
        
        pool_output = []
        for i,filter_size in enumerate(filter_sizes):
            with tf.name_scope("conv_pool-%s"%filter_size):

                fliter_shape = [filter_size,embedding_size,1,num_fliters]
                W = tf.Variable(tf.truncated_normal(fliter_shape,stddev=0.1))
                #W = tf.Variable(tf.contrib.layers.xavier_initializer()(fliter_shape))
                bias = tf.Variable(tf.constant(0.1,shape=[num_fliters]))

                conv = tf.nn.conv2d(input = self.embedding_chars_exp,filter = W,strides=[1,1,1,1],padding="VALID")
                #conv = tf.nn.conv2d(input = self.embedding_chars,filter = W,strides=[1,1,1,1],padding="VALID")
                #h = tf.nn.relu(tf.nn.bias_add(conv,bias))
                h = tf.nn.tanh(tf.nn.bias_add(conv,bias))

                pooling = tf.nn.max_pool(h, ksize = [1, sequence_length - filter_size + 1, 1, 1], strides = [1,1,1,1], padding = "VALID")
                pool_output.append(pooling)
        
        num_fliters_total = num_fliters*len(filter_sizes)
        self.h_pool = tf.concat(pool_output, 3)
        self.h_pool_flat = tf.reshape(self.h_pool,[-1,num_fliters_total])

        with tf.name_scope("dropout"):
            self.h_drop = tf.nn.dropout(self.h_pool_flat,self.drop_prob)
        
        # with tf.name_scope("full_connect"):
        #     num_hidden = 128
        #     W = tf.Variable(tf.truncated_normal([num_fliters_total, num_hidden],stddev=0.1))
        #     b = tf.Variable(tf.constant(0.1,shape=[num_hidden]))
        #     self.hidden_out = tf.nn.relu(tf.nn.xw_plus_b(self.h_drop,W,b))

        with tf.name_scope("out_put"):
            #W = tf.get_variable("W",shape=[num_fliters_total, num_class],initializer = tf.contrib.layers.xavier_initializer())
            #W = tf.Variable(tf.contrib.layers.xavier_initializer()([num_fliters_total, num_class]))
            W = tf.Variable(tf.truncated_normal([num_fliters_total, num_class],stddev=0.1))
            #W = tf.Variable(tf.truncated_normal([num_hidden, num_class],stddev=0.1))
            b = tf.Variable(tf.constant(0.1,shape=[num_class]))

            l2_loss += tf.nn.l2_loss(W)
            l2_loss += tf.nn.l2_loss(b)

            #self.# = tf.nn.sigmoid(tf.nn.xw_plus_b(self.h_drop,W,b))
            self.logits = tf.nn.xw_plus_b(self.h_drop,W,b)
            self.scores = tf.nn.softmax(self.logits)
            self.prediction = tf.arg_max(self.scores,1)
        
        with tf.name_scope("loss"):
            losses = tf.nn.softmax_cross_entropy_with_logits(logits=self.logits, labels=self.input_y)
            self.loss = tf.reduce_mean(losses) + l2_reg * l2_loss
            

        with tf.name_scope("acc"):
            correct_predictions = tf.equal(self.prediction, tf.arg_max(self.input_y, 1))
            trans_correct = tf.cast(correct_predictions,"float")

            self.correct_num = tf.reduce_sum(trans_correct)
            #self.accuracy = tf.reduce_mean(trans_correct)
            
def next_batch(data , batch_size, num):
    start_ind = batch_size*num
    end_ind = start_ind + batch_size

    if(end_ind > data.shape[0]):
        end_ind = data.shape[0]

    out = []
    for i in range(start_ind, end_ind):
        out.append(data[i])
    return out
    
print("load data...")
data = Data_pro()
train_x = data.x_train_ids
train_y = data.y_train

test_x = data.x_test_ids

W_list1 = data.wlist
#test_qid, test_text = data.read_test_csv()
test_qid = data.qid_test
K = 4

bool_train = True
sentence_max_len = data.maxlen
embedding_size = 300
filter_sizess = [1,2,3,5]
filter_numbers = 128
batch_sizes = 512
l2_reg_ = 0.0
keep_prob = 0.9
ecohs = 2
Acc=[]


cnn = TextCNN(W_list = W_list1, embd_train_bool=bool_train, sequence_length=sentence_max_len, num_class=2, embedding_size=embedding_size, filter_sizes=filter_sizess, num_fliters=filter_numbers, batch_size=batch_sizes, l2_reg = l2_reg_)

train_setup = tf.train.AdamOptimizer().minimize(cnn.loss)
total_batch_train = int((len(train_x) / batch_sizes))
total_batch_test = int((len(test_x) / batch_sizes))

if((len(train_x) % batch_sizes) != 0):
    total_batch_train += 1
    
if((len(test_x) % batch_sizes) != 0):
    total_batch_test += 1


sess = tf.Session()
sess.run(tf.global_variables_initializer())
print("Starting Training...")
for i in range(ecohs):
    
    for j in range(total_batch_train):
        batch_xs = next_batch(train_x,batch_sizes,j)
        batch_ys = next_batch(train_y,batch_sizes,j)

        sess.run(train_setup,feed_dict={cnn.input_x:batch_xs, cnn.input_y:batch_ys, cnn.drop_prob:keep_prob})
        if j % 100 == 0:
            print("Ecoh: " + str(i) + " Batch: " + str(j) + " TrainLoss: " + str(sess.run(cnn.loss, feed_dict={cnn.input_x:batch_xs, cnn.input_y:batch_ys, cnn.drop_prob:keep_prob})))
    
predict = []
out = []
for j in range(total_batch_test):
    batch_xs = next_batch(test_x,batch_sizes,j)
    predict.extend((sess.run(cnn.prediction, feed_dict={cnn.input_x:batch_xs, cnn.drop_prob:1.0})).tolist())

for i in predict:
    if(i == 1):
        out.append("0")#good
    else:
        out.append("1")#bad

dataframe = pd.DataFrame({'qid': test_qid, 'prediction':out})
dataframe.to_csv("submission.csv",index=False,sep=',')








