############################################
library(h2o)
library(data.table)
library(Metrics)
h2o.init(nthreads=-1)

library(dplyr)
library(data.table)
library(h2o)

cat("Read in the variables of the train and test datasets...\n")
cols <- c("Id", "radardist_km", "minutes_past", 
          "Ref", "RefComposite")
train <- fread("../input/train.csv", select = c(cols, "Expected"))
test <- fread("../input/test.csv", select = cols)
test$Expected <- c(0)

cat("Remove the highly skewed right tail...\n")
quantile(train$Expected, c(0.90, 0.95, 0.97, 0.98, 0.99))
train$Expected <- ifelse(train$Expected >= 50, 50, train$Expected)

# Get the time differences between each measure
time_difference <- function(times, num_per_segment = 60) {
  n <- length(times)
  valid_time <- vector(mode="numeric", length = n)
  valid_time[1] <- times[1]
  valid_time[-1] <- diff(times, 1)
  valid_time[n] <- valid_time[n] + num_per_segment - sum(valid_time)
  valid_time <- valid_time / num_per_segment
  valid_time
}

# Convert reflectivity (dbz) to mm/hr
marshall_palmer <- function(dbz) {
  ((10**(dbz/10))/200) ** 0.625
}

cat("Collapse the data according to Id...\n")
train$timeDiff <- time_difference(train$minutes_past)
train$mp <- marshall_palmer(train$Ref)
trainSum <- train %>% 
			group_by(Id) %>% 
			summarise(
				Expected = log1p(mean(Expected)), 
				Ref = mean(timeDiff * Ref), 
				RC = mean(timeDiff * RefComposite),
				mpRef = sum(timeDiff * Ref), 
				mpRC = sum(timeDiff * RefComposite), 
				maxRef = max(Ref),
				maxRC = max(RefComposite),
				rd = mean(radardist_km),
				records = n())
trainSum <- data.frame(trainSum)
test$timeDiff <- time_difference(test$minutes_past)
test$mp <- marshall_palmer(test$Ref)
testSum <- test %>% 
			group_by(Id) %>% 
			summarise(
				Ref = mean(timeDiff * Ref), 
				RC = mean(timeDiff * RefComposite),
				mpRef = sum(timeDiff * Ref), 
				mpRC = sum(timeDiff * RefComposite), 
				maxRef = max(Ref),
				maxRC = max(RefComposite),
				rd = mean(radardist_km),
				records = n())
testSum <- data.frame(testSum)
Id <- testSum$Id


cat("training model...\n")
trainHex <- as.h2o(trainSum, destination_frame="train.hex")
rfHex<-h2o.randomForest(x = c("Ref", "RC", "mpRef", "mpRC", "maxRef", "maxRC", "rd", "records"),
                        y="Expected",
                        training_frame=trainHex,
                        model_id="rfStarter.hex", 
                        ntrees=500, 
                        max_depth= 21)
rfHex

cat("Processing test data...\n")    
testHex <- as.h2o(testSum,destination_frame = "test.hex")
rf_prediction <- expm1(as.data.frame(h2o.predict(rfHex,testHex)))

sample_solution <- fread("../input/sample_solution.csv")
rf_solution <- data.frame(Id = testSum$Id, pred = rf_prediction$predict)
solution <- merge(sample_solution, rf_solution, by = "Id", all=TRUE)
solution$pred <- ifelse(is.na(solution$pred), 0, solution$pred)

cat("Generating the submission file...\n")    
sol <- data.frame(Id = solution$Id, 
                  Expected = solution$pred*0.8 + sample_solution$Expected*0.2)
write.csv(sol, "submission_RF.csv", row.names = FALSE)