rm(list=ls())
library(keras)
library(tidyverse)
library(qdapRegex)
library(data.table)
library(tensorflow)
library(tidytext)
library(stopwords)



# Read data, make validation set
train_data = read_csv("../input/train.csv")
set.seed(1992); val_inds = sample(1:nrow(train_data), floor(nrow(train_data)*0.08),F)
val_data = train_data[val_inds,]
train_data = train_data[-val_inds,]
test_data = read_csv("../input/test.csv")



#Record batch size, max words, other parameters
b_size = 512
max_words = 90*1000
maxl = 70
train.len <- nrow(train_data) #length of train set
val.len <- nrow(val_data) #length of validation set
test.len <- nrow(test_data) #length of test set
combined.data <- bind_rows(train_data, val_data,test_data); 
rm(train_data,val_data,test_data)



#VERY minor clean-up
combined.data$question_text <- str_to_lower(combined.data$question_text) %>%
  str_replace_all("\\d", " number ")



#Tokenize
wordseq = text_tokenizer(num_words = max_words) %>%
  fit_text_tokenizer(combined.data$question_text)
text2s = texts_to_sequences(wordseq, combined.data$question_text ) %>%
  pad_sequences( maxlen = maxl) 
word_index = wordseq$word_index
wordindex = unlist(wordseq$word_index)
rm(wordseq,val_inds); invisible(gc(reset=T))



#Split train/val/test sets
x_train1 = text2s[1:train.len,]
y_train = combined.data$target[1:train.len]
combined.data <- combined.data[-c(1:train.len),]; text2s <- text2s[-c(1:train.len),]

x_val1 = text2s[1:val.len,]
y_val = combined.data$target[1:val.len]
combined.data <- combined.data[-c(1:val.len),]; text2s <- text2s[-c(1:val.len),]

x_test1 = text2s[1:test.len,]
test_qid <- combined.data$qid
rm(combined.data,text2s)






#Run model for each of the three embbedings
embed.paths = list("../input/embeddings/wiki-news-300d-1M/wiki-news-300d-1M.vec",
                    "../input/embeddings/glove.840B.300d/glove.840B.300d.txt",
                    "../input/embeddings/paragram_300_sl999/paragram_300_sl999.txt")



for(embed.path in embed.paths){
    # this part re-purposed and modified some code from 
    # https://www.kaggle.com/mosnoiion/two-rnn-cnn-columns-networks-with-keras/notebook


    #Prepare embedding matrix:
    wgt = fread(embed.path, data.table = FALSE,skip=1)
    colnames(wgt)[1] <- "word"
    
    wgt = wgt %>%
    mutate(word=gsub("[[:punct:]]"," ", rm_white(word) ))
  
  
  dic_words = wgt$word
  dic = data.frame(word=as.character(names(wordindex)), key = wordindex,row.names = NULL) %>%
    arrange(key) %>% 
    .[1:max_words,]
  dic$word <- as.character(dic$word)
  
  w_embed = dic %>% 
    left_join(wgt)
  rm(wgt,dic); invisible(gc(reset=T))

  J = ncol(w_embed)
  ndim = J-2
  w_embed = w_embed [1:(max_words-1),3:J] %>%
    mutate_all(as.numeric) %>%
    mutate_all(round,6) %>%
    mutate_all(funs(replace(., is.na(.), 0))) 
  
  colnames(w_embed) = paste0("V",1:ndim)
  w_embed = rbind(rep(0, ndim), w_embed) %>%
    as.matrix()
  
  
  w_embed = list(array(w_embed , c(max_words, ndim)))
  
  
  
  #Model:
  inp1 = layer_input(shape = list(maxl))
  
  emm = inp1 %>%
    layer_embedding(input_dim = max_words, output_dim = ndim, input_length = maxl, weights = w_embed,trainable=TRUE) 
  
  model1 = emm %>%
    layer_spatial_dropout_1d(rate=0.1) %>%
    bidirectional(layer_cudnn_gru(units = 128, return_sequences = TRUE)) %>% 
    layer_batch_normalization() %>%
    layer_dropout(rate=0.1) 
  
  max_pool1 = model1 %>% layer_global_max_pooling_1d() 
  ave_pool1 = model1 %>% layer_global_average_pooling_1d()
  
  outp = layer_concatenate(list(ave_pool1, max_pool1)) %>%
    layer_dense(units = 16, activation = "relu") %>%
    layer_dense(units = 1, activation = "sigmoid")
  


  model = keras_model(list(inp1), outp)
  
  
  
  model %>% compile(
    optimizer = "adam",
    loss = "binary_crossentropy",
    metrics = "binary_accuracy"
  )



  # Fit with early stopping
  history = model %>% keras::fit(
    list(x_train1), y_train,
    epochs = 20,
    batch_size = b_size,
    validation_data = list(list(x_val1),y_val),
    verbose=1,
    shuffle=TRUE,
    callbacks = list(
      callback_early_stopping(monitor = "val_loss",patience = 1,
                              verbose = 1, mode = c( "min")),
      callback_csv_logger(paste0("logger-gru-",which(embed.paths==embed.path),".txt")),
      callback_model_checkpoint(paste0("quora_question_model-",which(embed.paths==embed.path),".h5"), save_best_only = TRUE)
    )
  )
  
  
  
  #Save history
  sink(paste0("history-gru-",which(embed.paths==embed.path),".txt")); print(history); sink()
  
  
  
  #Load best iteration of model
  model = load_model_hdf5(paste0("quora_question_model-",which(embed.paths==embed.path),".h5"))
  
  
  #Predict validation set
  pred = model %>%
    predict(list(x_val1), batch_size = b_size)
  pred = data.frame(prediction=pred)
  write_csv(pred,paste0(which(embed.paths==embed.path),"-validpart"))
  
  
  
  #Predict test set
  pred = model %>%
    predict(list(x_test1), batch_size = b_size)
  pred = data.frame(qid=test_qid, prediction=pred)
  write_csv(pred,paste0(which(embed.paths==embed.path),"-testpart"))
  
  
  #Free up some memoty
  rm(pred,w_embed,model,history); invisible(gc(reset=T))
}





#Sams as above but with lstm instead of gru
for(embed.path in embed.paths){
    # this part re-purposed and modified some code from 
    # https://www.kaggle.com/mosnoiion/two-rnn-cnn-columns-networks-with-keras/notebook


    #Prepare embedding matrix:
    wgt = fread(embed.path, data.table = FALSE,skip=1)
    colnames(wgt)[1] <- "word"
    
    wgt = wgt %>%
    mutate(word=gsub("[[:punct:]]"," ", rm_white(word) ))
  
  
  dic_words = wgt$word
  dic = data.frame(word=as.character(names(wordindex)), key = wordindex,row.names = NULL) %>%
    arrange(key) %>% 
    .[1:max_words,]
  dic$word <- as.character(dic$word)
  
  w_embed = dic %>% 
    left_join(wgt)
  rm(wgt,dic); invisible(gc(reset=T))

  J = ncol(w_embed)
  ndim = J-2
  w_embed = w_embed [1:(max_words-1),3:J] %>%
    mutate_all(as.numeric) %>%
    mutate_all(round,6) %>%
    mutate_all(funs(replace(., is.na(.), 0))) 
  
  colnames(w_embed) = paste0("V",1:ndim)
  w_embed = rbind(rep(0, ndim), w_embed) %>%
    as.matrix()
  
  
  w_embed = list(array(w_embed , c(max_words, ndim)))
  
  
  
  #Model:
  inp1 = layer_input(shape = list(maxl))
  
  emm = inp1 %>%
    layer_embedding(input_dim = max_words, output_dim = ndim, input_length = maxl, weights = w_embed,trainable=TRUE) 
  
  model1 = emm %>%
    layer_spatial_dropout_1d(rate=0.1) %>%
    bidirectional(layer_cudnn_lstm(units = 128, return_sequences = TRUE)) %>% 
    layer_batch_normalization() %>%
    layer_dropout(rate=0.1) 
  
  max_pool1 = model1 %>% layer_global_max_pooling_1d() 
  ave_pool1 = model1 %>% layer_global_average_pooling_1d()
  
  outp = layer_concatenate(list(ave_pool1, max_pool1)) %>%
    layer_dense(units = 16, activation = "relu") %>%
    layer_dense(units = 1, activation = "sigmoid")
  


  model = keras_model(list(inp1), outp)
  
  
  
  model %>% compile(
    optimizer = "adam",
    loss = "binary_crossentropy",
    metrics = "binary_accuracy"
  )



  # Fit with early stopping
  history = model %>% keras::fit(
    list(x_train1), y_train,
    epochs = 20,
    batch_size = b_size,
    validation_data = list(list(x_val1),y_val),
    verbose=1,
    shuffle=TRUE,
    callbacks = list(
      callback_early_stopping(monitor = "val_loss",patience = 1,
                              verbose = 1, mode = c( "min")),
      callback_csv_logger(paste0("logger-lstm-",which(embed.paths==embed.path),".txt")),
      callback_model_checkpoint(paste0("quora_question_model2-",which(embed.paths==embed.path),".h5"), save_best_only = TRUE)
    )
  )
  
  
  
  #Save history
  sink(paste0("history-lstm-",which(embed.paths==embed.path),".txt")); print(history); sink()
  
  
  
  #Load best iteration of model
  model = load_model_hdf5(paste0("quora_question_model2-",which(embed.paths==embed.path),".h5"))
  
  
  #Predict validation set
  pred = model %>%
    predict(list(x_val1), batch_size = b_size)
  pred = data.frame(prediction=pred)
  write_csv(pred,paste0(which(embed.paths==embed.path),"-validpart2"))
  
  
  
  #Predict test set
  pred = model %>%
    predict(list(x_test1), batch_size = b_size)
  pred = data.frame(qid=test_qid, prediction=pred)
  write_csv(pred,paste0(which(embed.paths==embed.path),"-testpart2"))
  
  
  #Free up some memory
  rm(pred,w_embed,model,history); invisible(gc(reset=T))
}


#Free up some memory
rm(x_train1,y_train,x_test1,x_val1)



#Determine threshold from validation set
pred <- data.frame(qid=1:length(y_val))
for(embed.path in embed.paths){
  pred.sub <- bind_cols(p1=read_csv(paste0(which(embed.paths==embed.path),"-validpart"))$prediction,
                        p2=read_csv(paste0(which(embed.paths==embed.path),"-validpart2"))$prediction)
                        
  colnames(pred.sub) <- paste0(c("gru.","lstm."),which(embed.paths==embed.path))                        
  pred <- bind_cols(pred,pred.sub)
  rm(pred.sub)
}

pred <- data.frame(prediction=rowMeans(pred[,2:ncol(pred)]),truth=y_val)



#Threshold search
sink(paste0("thresh-lstm-gru",".txt"))
  best.f1 <- best.thresh <- 0
  for(thresh in seq(0.2,0.5,0.01)){
    preds.thresh <- ifelse(pred$prediction >= thresh, 1, 0)
    y_val <- pred$truth
    Precision <- caret::precision(data = factor(preds.thresh), reference = factor(y_val), relevant = "1")
    Recall <- caret::recall(data = factor(preds.thresh), reference = factor(y_val), relevant = "1")
    f1.thresh <- 2 * (Precision * Recall)/(Precision + Recall)
    cat("Thresh = ", thresh, "     ","F1 = ", f1.thresh,"\n")
    best.thresh <- ifelse(f1.thresh >= best.f1, thresh,best.thresh)
    best.f1 <- ifelse(f1.thresh >= best.f1, f1.thresh,best.f1)
  }
sink()








#Predict TEST
pred <- data.frame(qid=read_csv(paste0(which(embed.paths==embed.path),"-testpart"))$qid)
for(embed.path in embed.paths){
  pred.sub <- bind_cols(p1=read_csv(paste0(which(embed.paths==embed.path),"-testpart"))$prediction,
                        p2=read_csv(paste0(which(embed.paths==embed.path),"-testpart2"))$prediction)
                        
  colnames(pred.sub) <- paste0(c("gru.","lstm."),which(embed.paths==embed.path))                        
  pred <- bind_cols(pred,pred.sub)
  rm(pred.sub)
}

pred <- data.frame(qid=pred$qid,prediction=rowMeans(pred[,2:ncol(pred)]))
pred$prediction <- ifelse(pred$prediction >= best.thresh, 1, 0)
write_csv(pred,paste0("submission.csv"))