############################################
## Use H2O to create a random forest
##  against the entire data set in 
##  just a couple minutes
##
## This is a starter script, using defaults
##  so it can be improved. And RF may not be
##  the best algorithm for this problem. But
##  the script shows that it can be done in R
##  fairly quickly, in fact. And it will scale
##  well to adding many more columns.
##
## To better fit MAE, the log of the 
##  target has been used: log1p/expm1
##
## The final step blends with the Marshall-Palmer
##  benchmark 50/50.
##
## The API for h2o.randomForest is shown 
##  at the bottom of the script
###########################################

library(h2o)
library(data.table)
library(Metrics)
h2o.init(nthreads=-1)

## use data table to only read the Estimated, Ref, and Id fields
print(paste("reading training file:",Sys.time()))
train<-fread("../input/train.csv",select=c(1,2,3,4,5,6,7,8,9,10,11,24))

#Examin outliers of Expected >= 70
outtrain <- subset(train, Expected >= 70)
summary(outtrain)

#Cut off outliers of Expected >= 70
train <- subset(train, Expected < 70)

summary(train)

#Cut off Ref values < 0
for (i in 1:ncol(train)) {
train[,i][which(train[,i]< 0)] <- NA
}


summary(train)

