# This R environment comes with all of CRAN preinstalled, as well as many other helpful packages
# The environment is defined by the kaggle/rstats docker image: https://github.com/kaggle/docker-rstats
# For example, here's several helpful packages to load in 

library(ggplot2) # Data visualization
library(readr) # CSV file I/O, e.g. the read_csv function

# Input data files are available in the "../input/" directory.
# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory

system("ls ../input")

# Any results you write to the current directory are saved as output.

library(data.table)
#setwd("e:\\MyWork\\KHD2016\\projects\\expedia")
expedia_train <- fread('../input/train.csv', header=TRUE)
expedia_test <- fread('../input/test.csv', header=TRUE)

dest_id_hotel_cluster_count <-
expedia_train[,length(is_booking),
by=list(srch_destination_id, hotel_cluster)]

top_five <- function(hc,v1) {
	hc_sorted <- hc[order(v1,decreasing=TRUE)]
	n <- min(5,length(-hc_sorted))
	paste(hc_sorted[1:n],collapse=" ")
}

dest_top_five <-dest_id_hotel_cluster_count[,top_five(hotel_cluster,V1),by=srch_destination_id]

dd <- merge(expedia_test,dest_top_five,by="srch_destination_id",all.x=TRUE)[order(id),list(id,V1)]
setnames(dd,c("id","hotel_cluster"))
write.csv(dd, file='submission.csv', row.names=FALSE)