# H2O Gradient Boost Machine Starter Script
# Based on Ben Hamner script from Springleaf
# https://www.kaggle.com/benhamner/springleaf-marketing-response/random-forest-example
# Forked from Random Forest Example by Michael Pawlus

#  Trying out h2o.gbm()

library(data.table)  
library(h2o)

cat("reading the train and test data (with data.table) \n")
train <- fread("../input/train.csv",stringsAsFactors = T)
test  <- fread("../input/test.csv",stringsAsFactors = T)
store <- fread("../input/store.csv",stringsAsFactors = T)
train <- train[Sales > 0,]  ## We are not judged on 0 sales records in test set
    ## See Scripts discussion from 10/8 for more explanation.
train <- merge(train,store,by="Store")
test <- merge(test,store,by="Store")

cat("train data column names and details\n")
summary(train)
cat("test data column names and details\n")
summary(test)

## more care should be taken to ensure the dates of test can be projected from train
## decision trees do not project well, so you will want to have some strategy here, if using the dates
train[,Date:=as.Date(Date)]
test[,Date:=as.Date(Date)]

# seperating out the elements of the date column for the train set
train[,month:=as.integer(format(Date, "%m"))]
train[,year:=as.integer(format(Date, "%y"))]
train[,Store:=as.factor(as.numeric(Store))]

test[,month:=as.integer(format(Date, "%m"))]
test[,year:=as.integer(format(Date, "%y"))]
test[,Store:=as.factor(as.numeric(Store))]

## log transformation to not be as sensitive to high sales
## decent rule of thumb: 
##     if the data spans an order of magnitude, consider a log transform
train[,logSales:=log1p(Sales)]

## Use H2O's gradient boost machine 
## Start cluster with all available threads
localH2O = h2o.init(nthreads=-1,max_mem_size='6G')
## Load data into cluster from R
print("Loading data into the cluster from R")
trainHex<-as.h2o(train, localH2O)
## Set up variable to use all features other than those specified here
print("setting up independent variables")
features<-colnames(train)[!(colnames(train) %in% c("Id","Date","Sales","logSales","Customers"))]


## Train a gradient boosting machine using all default parameters
print("training gbm model")
gbmHex <- h2o.gbm(x=features,
                  y="logSales",
                  distribution="gaussian",
                  training_frame=trainHex,
                  n.trees = 100,
                  interaction.depth = 5,
                  n.minobsinnode = 2,
                  shrinkage = 0.2,
                  n.bins = 1115 ## allow it to fit store ID
                  )

summary(gbmHex)
cat("Predicting Sales\n")
## Load test data into cluster from R
testHex<-as.h2o(test)

## Get predictions out; predicts in H2O, as.data.frame gets them into R
predictions<-as.data.frame(h2o.predict(gbmHex,testHex))
## Return the predictions to the original scale of the Sales data
pred <- expm1(predictions[,1])
summary(pred)
submission <- data.frame(Id=test$Id, Sales=pred)

cat("saving the submission file\n")
write.csv(submission, "h2o_gbm.csv",row.names=F)





# http://www.rdocumentation.org/packages/h2o/functions/h2o.gbm

# Arguments
# x = A vector containing the names or indices of the predictor variables to use in building the GBM model.
# y = The name or index of the response variable. If the data does not contain a header, this is the column index number starting at 0, and increasing from left to right. (The response must be either an integer or a categorical variable).
# distribution = The type of GBM model to be produced: classification is "multinomial" (default), "gaussian" is used for regression.
# data = An H2OParsedData object containing the variables in the model.
# key = (Optional) The unique hex key assigned to the resulting model. If none is given, a key will automatically be generated.
# n.trees = (Optional) Number of trees to grow. Must be a nonnegative integer.
# interaction.depth = (Optional) Maximum depth to grow the tree.
# n.minobsinnode = (Optional) Minimum number of rows to assign to teminal nodes.
# shrinkage = (Optional) A learning-rate parameter defining step size reduction.
# n.bins = (Optional) Number of bins to use in building histogram.
# importance = (Optional) A logical value indicating whether variable importance should be calculated. This will increase the amount of time for the algorithm to complete.
# nfolds = (Optional) Number of folds for cross-validation. If nfolds >= 2, then validation must remain empty.
# validation = (Optional) An H2OParsedData object indicating the validation dataset used to construct confusion matrix. If left blank, this defaults to the training data when nfolds = 0.
# balance.classes = (Optional) Balance training data class counts via over/under-sampling (for imbalanced data)
# max.after.balance.size = Maximum relative size of the training data after balancing class counts (can be less than 1.0)


# Value
# An object of class H2OGBMModel with slots key, data, valid (the validation dataset) and model, where the last is a list of the following components:
# type = The type of the tree.
# n.trees = Number of trees grown.
# oob_err = Out of bag error rate.
# forest = A matrix giving the minimum, mean, and maximum of the tree depth and number of leaves.
# confusion = Confusion matrix of the prediction when classification model is specified.