# This R environment comes with all of CRAN preinstalled, as well as many other helpful packages
# The environment is defined by the kaggle/rstats docker image: https://github.com/kaggle/docker-rstats
# For example, here's several helpful packages to load in 

library(ggplot2) # Data visualization
library(readr) # CSV file I/O, e.g. the read_csv function
library(quanteda)
library(data.table)
library(randomForest)
library(e1071)
library(dplyr)
library(qdap)
library(proxy)
library(parallel)
library(tm)
# Input data files are available in the "../input/" directory.
# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory

system("ls ../input")

#tr <- read_delim('../input/train.csv', delim=',')
te <- read_delim('../input/test.csv', delim=',')

ProcessQs <- function(q){
  w <-  toLower(
    tokenize(
      as.character(q), 
      what = 'word', 
      removeNumbers = T, 
      removePunct = T, 
      removeSymbols = T, 
      removeSeparators = T, 
      removeTwitter = T, 
      removeHyphens = T, 
      removeURL = T)
  )
  
  w <- as.data.frame(
    as.matrix(
      DocumentTermMatrix(
        SimpleCorpus(
          VectorSource(w)))))
  
  if(length(w) < 1){
    ret <- data.frame(fake_entry = 1)   
  }
  else{
    ret <- w
  }
  
  return(ret)
}


ProcessQs(te$question1[1])

CompareQs <- function(k){
  w <- rbindlist(list(ProcessQs(te$question1[k]), 
                      ProcessQs(te$question2[k])), fill = T)
  w <- w %>%
    mutate_all(function(x) ifelse(is.na(x) == TRUE,0,x))
  
  print(paste(Sys.time(), k, sep = "  _"))
  return(data.frame(test_id = te$test_id[k],
                    is_duplicate = ifelse(simil(w, method = 'cosine') > .6*ncol(w)/ncol(w), 1, 0)))
}


# sample
p <- as.data.frame(do.call(rbind, Map(CompareQs, 1:1000)))
write.csv(p,'submission.csv',row.names=FALSE)


# Any results you write to the current directory are saved as output.