library(dplyr)
library(data.table)
library(xgboost)

cat("Read in the variables of the train and test datasets...\n")
cols <- c("Id", "radardist_km", "minutes_past", 
          "Ref", "RefComposite")
train <- fread("../input/train.csv", select = c(cols, "Expected"))
test <- fread("../input/test.csv", select = cols)
test$Expected <- c(0)

cat("Remove the highly skewed right tail...\n")
quantile(train$Expected, c(0.90, 0.95, 0.97, 0.98, 0.99))
train$Expected <- ifelse(train$Expected >= 50, 50, train$Expected)

# Get the time differences between each measure
time_difference <- function(times, num_per_segment = 60) {
  n <- length(times)
  valid_time <- vector(mode="numeric", length = n)
  valid_time[1] <- times[1]
  valid_time[-1] <- diff(times, 1)
  valid_time[n] <- valid_time[n] + num_per_segment - sum(valid_time)
  valid_time <- valid_time / num_per_segment
  valid_time
}

# Convert reflectivity (dbz) to mm/hr
marshall_palmer <- function(dbz) {
  ((10**(dbz/10))/200) ** 0.625
}

cat("Collapse the data according to Id...\n")
train$timeDiff <- time_difference(train$minutes_past)
train$mp <- marshall_palmer(train$Ref)
trainSum <- train %>% 
			group_by(Id) %>% 
			summarise(
				Expected = log1p(mean(Expected)), 
				Ref = mean(timeDiff * Ref), 
				RC = mean(timeDiff * RefComposite),
				mpRef = sum(timeDiff * Ref), 
				mpRC = sum(timeDiff * RefComposite), 
				maxRef = max(Ref),
				maxRC = max(RefComposite),
				rd = mean(radardist_km),
				records = n())
trainSum <- data.frame(trainSum)
test$timeDiff <- time_difference(test$minutes_past)
test$mp <- marshall_palmer(test$Ref)
testSum <- test %>% 
			group_by(Id) %>% 
			summarise(
				Ref = mean(timeDiff * Ref), 
				RC = mean(timeDiff * RefComposite),
				mpRef = sum(timeDiff * Ref), 
				mpRC = sum(timeDiff * RefComposite), 
				maxRef = max(Ref),
				maxRC = max(RefComposite),
				rd = mean(radardist_km),
				records = n())
testSum <- data.frame(testSum)
Id <- testSum$Id

names(trainSum)
names(testSum)
				
cat("training model...\n")
Expected <- trainSum$Expected
trainSum <- as.matrix(trainSum[,-c(1:2)])
trainSum <- matrix(as.numeric(trainSum),nrow(trainSum),ncol(trainSum))
trainSum <- xgb.DMatrix(trainSum, label = Expected)

testSum <- as.matrix(testSum[, -1])
testSum <- matrix(as.numeric(testSum),nrow(testSum),ncol(testSum))
testSum <- xgb.DMatrix(testSum)


feature_cols <- c("Ref", "RC", "mpRef", "mpRC", "maxRef", "maxRC", "rd", "records")
xgbHex <- xgboost(data = trainSum, label = Expected,
                objective  = "reg:linear", 
                eval_metric = "rmse",
                eta = 0.015,
                subsample = 0.7,
                min_child_weight =10,    
                max_depth = 6,
                nthreads = 4,
                nrounds = 1000)
xgbHex

cat("Processing test data...\n")    
xgb_prediction <- expm1(predict(xgbHex,testSum))
sample_solution <- fread("../input/sample_solution.csv")
xgb_solution <- data.frame(Id = Id, pred = xgb_prediction)
solution <- merge(sample_solution, xgb_solution, by = "Id", all=TRUE)
solution$pred <- ifelse(is.na(solution$pred), 0, solution$pred)

cat("Generating the submission file...\n")    
sol <- data.frame(Id = solution$Id, 
                  Expected = solution$pred*0.8 + sample_solution$Expected*0.2)
write.csv(sol, "submission_XGB.csv", row.names = FALSE)