# This R environment comes with all of CRAN preinstalled, as well as many other helpful packages
# The environment is defined by the kaggle/rstats docker image: https://github.com/kaggle/docker-rstats
# For example, here's several helpful packages to load in 

# library(ggplot2) # Data visualization
library(readr) # CSV file I/O, e.g. the read_csv function

library(data.table)

# Input data files are available in the "../input/" directory.
# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory

# options(echo = FALSE) # ECHO OFF
print('#######################################################################################################')
print('# Outbrain Click Prediction - JJTZ 2016')
print('#######################################################################################################')

# Leer trainset:
trainset <- fread("../input/clicks_train.csv")
setkey(trainset, "ad_id")

# head(trainset)
print(str(trainset))
# summary(trainset)

# Proporción de clicks (global):
prob_click_global <- mean(trainset$clicked)

# Crear primer submit con las frecuencias como prob.:
probs_ads <- trainset[, .(prob = mean(clicked) ), by = ad_id]

# head(probs_ads)

# Leer testset:
testset <- fread( "../input/clicks_test.csv")
setkey(testset, "ad_id")
# head(testset)
print(str(testset))
# summary(testset)

prob_click_global_ads_no_en_test <- mean(trainset$clicked[!(trainset$ad_id %in% testset$ad_id)])
prob_click_global_ads_en_test <- mean(trainset$clicked[trainset$ad_id %in% testset$ad_id])

# # Free memory:
# rm(trainset)
# gc() # Garbage collector

# Añadir las probs de los ads:
testset <- merge(testset, probs_ads, all.x = T, by = "ad_id") # by ad_id

# Predecir los NAs (con prob_click_global_ads_en_test) :
testset[is.na(testset$prob), prob := prob_click_global_ads_en_test]

# summary(testset)

# PREPARACIÓN DEL SUBMIT:
# length(unique(testset$display_id)) == 6.245.533 rows

# Para conseguir los Ads de mayor prob a menor prob, se pone la "key" al revés:
testset[!is.na(testset$prob), prob := 1 - prob] # Ahora prob es la prob de "no click"!
# Ahora usamos ese orden para crear el submit:
setkey(testset, "prob")

print('Creando submitset...')
submitset <- testset[, .(ad_id = paste(ad_id, collapse=" ")), by = display_id] # 85 secs
#print(mi_tiempo['elapsed'])
print('Submitset ok!')

# head(submitset)

# Ordenamos por display_id:
setkey(submitset,"display_id")

print('Guardando fichero...')
mi_tiempo <- system.time({
  write.table(submitset, file = "submitset.csv", row.names = F, quote = FALSE, sep = ",")
})
print(mi_tiempo['elapsed'])

print('Ok.')
# Any results you write to the current directory are saved as output.