# This R environment comes with all of CRAN preinstalled, as well as many other helpful packages
# The environment is defined by the kaggle/rstats docker image: https://github.com/kaggle/docker-rstats
# For example, here's several helpful packages to load in 

library(ggplot2) # Data visualization
library(readr) # CSV file I/O, e.g. the read_csv function

# Input data files are available in the "../input/" directory.
# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory

system("ls ../input")

# Any results you write to the current directory are saved as output.

library(data.table)
train <-fread("../input/train.csv")
test <-fread("../input/test.csv")
dim(train)

# cbind(train,adist = diag(adist(train$question1,train$question2)))

library(stringi)
first <- stri_extract_first_words(train$question1)
second <- stri_extract_first_words(train$question2)
first1 = stri_extract_last_words(train$question1)
second1 = stri_extract_last_words(train$question2)

train$first = ifelse(first == second, 1, 0 )
train$second = ifelse(first1 == second1, 1, 0 )

library(syuzhet)
sentiment1 <- get_nrc_sentiment(train$question1)
sentiment2 <- get_nrc_sentiment(train$question2)


train = cbind(train,sentiment1,sentiment2)
tr = train[,6:28]
class(tr)
tr[,c(1:23)] = lapply(tr[,c(1:23)],as.factor)
str(tr)
tr = na.omit(tr)

library(randomForest)
model <- randomForest(is_duplicate ~ ., data = tr)
model

pred <- predict(model, newdata = tr,  type="prob")

#bound the results, otherwise you might get infinity results
pred = apply(pred, c(1,2), function(x) min(max(x, 1E-15), 1-1E-15)) 

logLoss = function(pred, actual){
  -1*mean(log(pred[model.matrix(~ actual + 0) - pred > 0]))
}


### Logloss value = 
logLoss(pred, tr$is_duplicate)

write.csv(pred[,1],"words_benchmark.csv")
