## I'm so sorry about my English skill... I'm not good at English

## I'll not give you output. Because 'H2o' Package that useful DeepLeaning analysis tool
## must link each Computer's java programe and R programe. So I couldn't display output.

## Importing packages
#install.packages("readr")
#install.packages("h2o")
#library(readr)
#library(h2o)
## These libraries are my favorite R libraries. There are include many functions in h2o.
#
### Running code
#bat <- read_csv("C:\\Users\\LGPC\\Documents\\train_V2.csv")
#bat <- as.data.frame(bat)
#rank <- bat[, c(8, 16, 25)] ; rank[1:20 , ]
#summary(rank[rank$killPoints > 0, ]$killPoints)
#summary(rank[rank$rankPoints > 0, ]$rankPoints)
#summary(rank[rank$winPoints > 0, ]$winPoints)
#rank <- ifelse(rank$rankPoints > 0, rank$rankPoints, (rank$winPoints - 16))
#quantile(rank, probs = seq(0, 1, .1)) ; quantile(rank, probs = c(.005, .995)) 
#
## Points merge
#bat$rank <- rank ; bat$killPoints = NULL ; bat$winPoints = NULL ; bat$rankPoints = NULL 
## use .5%  ~  99.5% data. 
#bat <- bat[(bat$rank >= 1245) & (bat$rank <= 1837), ] 
#
## delete outlier (MatchDuration)
#quantile(bat$matchDuration, probs = seq(0, .1, .005)) 
#bat <- bat[bat$matchDuration >= 1167, ] 
#ng <- bat$numGroups ; summary(ng) ; quantile(ng, probs = seq(0, .1, .005)) 
## delete 'numGruops' under 16
#bat <- bat[bat$numGroups >= 16, ]; ng <- ng[ng >= 16]
#bat$maxPlace = NULL ; bat$numGroups = NULL
#
## k-Means clustering about match Type. biggest number of group is Solo data
#set.seed(123)
#ng_km <- kmeans(ng, center = 3, iter.max = 7000) ; ng_km  # duo:1 , solo:2 , squad:3
#MT <- bat$matchType
#
#bat$per <- bat$winPlacePerc ; bat$winPlacePerc = NULL ; bat$matchType = NULL
#
## delete unuseful feature
#full.fit <- lm(per ~ ., data = bat) ; summary(full.fit) ; vif(full.fit)
#bat$kills = NULL
#full.fit <- lm(per ~ ., data = bat) ; summary(full.fit) ; vif(full.fit)
#bat$headshotKills = NULL ; bat$roadKills = NULL ; bat$vehicleDestroys = NULL ; 
#
## K-Means clustering about matchDuration. We can splite matchDuration into BattleRoyal and MiniRoyal.
#md <- bat$matchDuration
#set.seed(123) ; kmd <- kmeans(md, centers = 2, iter.max = 7000) ; bat$matchDuration <- NULL
## bat : 1 , mini : 2
#bat$bat_or_mini <- kmd$cluster 
#
#solo$assists = 0 ; solo$DBNOs = 0 ; solo$revives = 0 ; solo$teamKills = 0
#
#bom <- solo$bat_or_mini ; solo$bat_or_mini = NULL 
#bat <- solo[bom == 1, ] ; mini <- solo[bom == 2, ]
#
## Use H2o Package. 
#solo <- as.h2o(solo) ; bat <- as.h2o(bat) ; mini <- as.h2o(mini)
#rand <- h2o.runif(solo, seed = 123)
#solotr <- solo[rand <= .7, ] ; solote <- solo[rand > .7, ]
#rand <- h2o.runif(bat, seed = 123)
#battr <- bat[rand <= .7, ] ; batte <- bat[rand > .7, ]
#rand <- h2o.runif(mini, seed = 123)
#minitr <- mini[rand <= .7, ] ; minite <- mini[rand > .7, ]
#
### Use RandomForest Model
#b.rf.model <- h2o.randomForest(x = 1:15,
#                               y = 16,
#                               training_frame = battr,
#                               validation_frame = batte, 
#                               model_id = "b.rf.model",
#                               distribution = "gaussian",
#                               ntree = 175)
#m.rf.model <- h2o.randomForest(x = 1:15, 
#                               y = 16,
#                               training_frame = minitr,
#                               validation_frame = minite,
#                               model_id = "m.rf.model",
#                               distribution = "gaussian",
#                               ntree = 175)
#a1 <- h2o.predict(b.rf.model, batte)
#a1 <- ifelse(a1 > 1, 1, a1)
#a1 <- ifelse(a1 < 0, 0, a1)
#b.rf.res <- a1 - batte$per
## Solo BattleRoyal's MSE : .00652
#mean(b.rf.res^2) 
#a2 <- h2o.predict(m.rf.model, minite)
#a2 <- ifelse(a2 > 1, 1, a2)
#a2 <- ifelse(a2 < 0, 0, a2)
#m.rf.res <- a2 - minite$per
## Solo MiniRoyal's MSE : .00603
#mean(m.rf.res^2) 
#
### Use generalized boosted regression Model
#b.gbm.model <- h2o.gbm(x = 1:15,
#                       y = 16,
#                       training_frame = battr,
#                       validation_frame = batte, 
#                       nfolds = 5,
#                       fold_assignment = "Random",
#                       ntrees = 175,
#                       seed = 123)
#m.gbm.model <- h2o.gbm(x = 1:15,
#                       y = 16,
#                       training_frame = minitr,
#                       validation_frame = minite,
#                       fold_assignment ="Random",
#                       distribution = "gaussian",
#                       ntrees = 175,
#                       nfolds = 5,
#                       seed = 123)
#a1 <- h2o.predict(b.gbm.model, batte)
#a1 <- ifelse(a1 > 1, 1, a1)
#a1 <- ifelse(a1 < 0, 0, a1)
#b.gbm.res <- a1 - batte$per
## .004557565
#mean(b.gbm.res^2) 
#a2 <- h2o.predict(m.gbm.model, minite)
#a2 <- ifelse(a2 > 1, 1, a2)
#a2 <- ifelse(a2 < 0, 0, a2)
#m.gbm.res <- a2 - minite$per
## .003985434
#mean(m.gbm.res^2) 
#
### So I decided Use H2o's GBM Modeling
### I made all about model solo, duo, squad, flare, crash.
### But flare and crash type's MSE are low when I use RandomForest Model.
#solobat.model <- h2o.gbm(x = 1:15,
#                         y = 16,
#                         training_frame = bat,
#                         nfolds = 5,
#                         fold_assignment = "Random",
#                         distribution = "gaussian",
#                         ntrees = 175,
#                         seed = 123)
#solomini.model <- h2o.gbm(x = 1:15,
#                          y = 16,
#                          training_frame = mini,
#                          nfolds = 5,
#                          fold_assignment = "Random",
#                          distribution = "gaussian",
#                          ntrees = 175,
#                          seed = 123)
#
### Duo
#duo <- rbind(Bat[MT == "duo-fpp",], Bat[MT == "duo",])
#
#md <- duo$matchDuration
#set.seed(123) ; kmd <- kmeans(md, centers = 2, iter.max = 7000) ; duo$matchDuration <- NULL
#duo$Bat_or_mini <- kmd$cluster # Bat : 2 , mini : 1
#
### I change data. If groupId same data, change data with feature's mean.
### Because winPlacePerc is same each team. Solo data isn't unnecessary this work.
#
#a <- duo$groupId
#b <- factor(a, levels <- levels(factor(a)), order = T)
#c <- as.numeric(b) ; duo$gi <- c ; duo$groupId = NULL
#
#f1 <- aggregate(duo$assists, by = list(duo$gi), mean)
#f2 <- aggregate(duo$boosts, by = list(duo$gi), mean)
#f3 <- aggregate(duo$damageDealt, by = list(duo$gi), mean)
#f4 <- aggregate(duo$DBNOs, by = list(duo$gi), mean)
#f5 <- aggregate(duo$heals, by = list(duo$gi), mean)
#f6 <- aggregate(duo$killPlace, by = list(duo$gi), mean)
#f7 <- aggregate(duo$killStreaks, by = list(duo$gi), mean)
#f8 <- aggregate(duo$longestKill, by = list(duo$gi), mean)
#f9 <- aggregate(duo$revives, by = list(duo$gi), mean)
#f10 <- aggregate(duo$rideDistance, by = list(duo$gi), mean)
#f11 <- aggregate(duo$swimDistance, by = list(duo$gi), mean)
#f12 <- aggregate(duo$teamKills, by = list(duo$gi), mean)
#f13 <- aggregate(duo$walkDistance, by = list(duo$gi), mean)
#f14 <- aggregate(duo$weaponsAcquired, by = list(duo$gi), mean)
#f15 <- aggregate(duo$rank, by = list(duo$gi), mean)
#
#duo <- duo[order(duo$gi), ]
#a1 <- merge(duo, f1, by.x = "gi", by.y = "Group.1", all = T)$x
#a2 <- merge(duo, f2, by.x = "gi", by.y = "Group.1", all = T)$x
#a3 <- merge(duo, f3, by.x = "gi", by.y = "Group.1", all = T)$x
#a4 <- merge(duo, f4, by.x = "gi", by.y = "Group.1", all = T)$x
#a5 <- merge(duo, f5, by.x = "gi", by.y = "Group.1", all = T)$x
#a6 <- merge(duo, f6, by.x = "gi", by.y = "Group.1", all = T)$x
#a7 <- merge(duo, f7, by.x = "gi", by.y = "Group.1", all = T)$x
#a8 <- merge(duo, f8, by.x = "gi", by.y = "Group.1", all = T)$x
#a9 <- merge(duo, f9, by.x = "gi", by.y = "Group.1", all = T)$x
#a10 <- merge(duo, f10, by.x = "gi", by.y = "Group.1", all = T)$x
#a11 <- merge(duo, f11, by.x = "gi", by.y = "Group.1", all = T)$x
#a12 <- merge(duo, f12, by.x = "gi", by.y = "Group.1", all = T)$x
#a13 <- merge(duo, f13, by.x = "gi", by.y = "Group.1", all = T)$x
#a14 <- merge(duo, f14, by.x = "gi", by.y = "Group.1", all = T)$x
#a15 <- merge(duo, f15, by.x = "gi", by.y = "Group.1", all = T)$x
#
#duo$assists <- a1 ; duo$boosts <- a2 ; duo$damageDealt <- a3 ; duo$DBNOs <- a4 ; duo$heals <- a5
#duo$killPlace <- a6 ; duo$killStreaks <- a7 ; duo$longestKill <- a8 ; duo$revives <- a9
#duo$rideDistance <- a10 ; duo$swimDistance <- a11 ; duo$teamKills <- a12 ; duo$walkDistance <- a13
#duo$weaponsAcquired <- a14 ; duo$rank <- a15
#duo <- duo[order(as.integer(rownames(duo))), ] ; duo <- duo[!duplicated(duo$gi), ]
#duo$gi <- NULL ; bom <- duo$Bat_or_mini ; duo$Bat_or_mini = NULL ; 
#bat <- duo[bom == 2, ] ; mini <- duo[bom == 1, ]
#
#
#duo <- as.h2o(duo) ; bat <- as.h2o(bat) ; mini <- as.h2o(mini)
#duobat.model <- h2o.gbm(x = 1:15,
#                        y = 16,
#                        training_frame = bat,
#                        nfolds = 5,
#                        fold_assignment = "Random",
#                        ntrees = 175,
#                        seed = 123)
#duomini.model <- h2o.gbm(x = 1:15,
#                         y = 16,
#                         training_frame = mini,
#                         fold_assignment ="Random",
#                         ntrees = 175,
#                         nfolds = 5,
#                         seed = 123)
#
### Squad, Flare and Crash Mode regressions are simliar duo data regrssion.
### So I'll pass those analysis.
#
### Result
#test <- read.csv("test_V2.csv")
#solo <- rbind(test[MT == "solo", ], test[MT == "solo-fpp", ], test[MT == "normal-solo", ], test[MT == "normal-solo-fpp", ])
#duo <- rbind(test[MT == "duo", ], test[MT == "duo-fpp", ], test[MT == "normal-duo", ], test[MT == "normal-duo-fpp", ])
#squad <- rbind(test[MT == "squad", ], test[MT == "squad-fpp", ], test[MT == "normal-squad", ], test[MT == "normal-squad-fpp", ])
#flare <- rbind(test[MT == "flarefpp", ], test[MT == "flaretpp", ]) 
#crash <- rbind(test[MT == "crashfpp", ], test[MT == "crashtpp", ]) 
#
#solobat <- as.h2o(solobat) ; solomini <- as.h2o(solomini) ; duobat <- as.h2o(duobat) ; duomini <- as.h2o(duomini)
#squadbat <- as.h2o(squadbat) ; squadmini <- as.h2o(squadmini) ; flare <- as.h2o(flare) ; crash <- as.h2o(crash)
#
#solobatpredict <- h2o.predict(solobat.model, solobat)
#solominipredict <- h2o.predict(solomini.model, solomini)
#duobatpredict <- h2o.predict(duobat.model, duobat)
#duominipredict <- h2o.predict(duomini.model, duomini)
#squadbatpredict <- h2o.predict(squadbat.model, squadbat)
#squadminipredict <- h2o.predict(squadmini.model, squadmini)
#flarepredict <- h2o.predict(flare.model, flare)
#crashpredict <- h2o.predict(crash.model, crash)
#
#solobat['per'] <- solobatpredict ; solomini['per'] <- solominipredict
#duobat['per'] <- duobatpredict ; duomini['per'] <- duominipredict
#squadbat['per'] <- squadbatpredict ; squadmini['per'] <- squadminipredict
#flare['per'] <- flarepredict ; crash['per'] <- crashpredict
#
#solobat <- as.data.frame(solobat) 
#solomini <- as.data.frame(solomini)
#duobat <- as.data.frame(duobat) 
#duomini <- as.data.frame(duomini) 
#squadbat <- as.data.frame(squadbat)
#squadmini <- as.data.frame(squadmini)
#flare <- as.data.frame(flare)
#crash <- as.data.frame(crash)
#
#final_test <- rbind(solobat, solomini, duobat, duomini, squadbat, squadmini, flare, crash)
#final_test <- final_test[order(final_test$nrow), ]
#per <- final_test$per
#per <- ifelse(per > 1, 1, per)
#per <- ifelse(per < 0, 0, per)
#final_test$per <- per
#
#sample_submission_V2 <- read.csv("sample_submission_V2.csv")
#sample_submission_V2$winPlacePerc <- per
#write.csv(sample_submission_V2,
#          file = "sample_submission_V2.csv",
#          row.names = T)
#          
sub <- read.csv("../input/my-submission/sample_submission_V2.csv")
sub$X = NULL
write.csv(sub, file = "sample_submission.csv", row.names = F)

# This work is first analysis in Kaggle. So I don't know how to submit...
# I was able to submit it..