# load the necessary packages 
require(data.table)

#load two submissions
xf1 <- fread('../submissions/submission1.csv')
xf2 <- fread('../submissions/submission2.csv')

# ensure they are ordered the same way :-)
xf1 <- xf1[order(xf1$click_id)]
xf2 <- xf2[order(xf2$click_id)]

# rank normalization
xf1$is_attributed <- rank(xf1$is_attributed)/nrow(xf1)
xf2$is_attributed <- rank(xf2$is_attributed)/nrow(xf2)

## arithmetic average
xfor <- xf1

# this one is a bit of a guess - normally I calculate the correlation
# between ranks along the lines of:
# alpha <- cor(xf1$is_attributed, xf2$is_attributed)
# and assign a higher weight to the better LB performer, except for situations 
# of comparable performance for two genuinely diverse models - in those cases
# equal weighting (alpha = 0.5) seems to be better. Obviously, if you do a proper 
# validation, those considerations are redundant :-) 

alpha <- 0.9
xfor$is_attributed <- alpha * xf1$is_attributed + (1-alpha) *  xf2$is_attributed
write.csv(xfor, '../submissions/arith_rank_blend.csv')

## geometric average - *usually* gives slightly superior results
xfor <- xf1
alpha <- 0.9
xfor$is_attributed <- exp(alpha * log(xf1$is_attributed) + (1-alpha) * log(xf2$is_attributed))
write.csv(xfor, '../submissions/geo_rank_blend.csv')


write.csv(xfor, '../submissions/combined_sub.csv', row.names = F, quote = F)
