# This R environment comes with all of CRAN preinstalled, as well as many other helpful packages
# The environment is defined by the kaggle/rstats docker image: https://github.com/kaggle/docker-rstats
# For example, here's several helpful packages to load in 

library(ggplot2) # Data visualization
library(readr) # CSV file I/O, e.g. the read_csv function

# Input data files are available in the "../input/" directory.
# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory

system("ls ../input")

# Any results you write to the current directory are saved as output.

library(data.table)

Yuqing_expedia_train <- fread('../input/train.csv', header=TRUE, 
select= c("is_booking","orig_destination_distance","hotel_cluster","srch_destination_id"))
# Yuqing_expedia_test <- fread('../input/test.csv', header=TRUE)

Yuqing_expedia_test <- fread('../input/test.csv', header=TRUE)

Trial_train = Yuqing_expedia_train[1:20,]
Trial_test = Yuqing_expedia_test[1:20,]

sum_and_count <- function(x){
  sum(x)*0.8456 + length(x) *(1-0.8456)
}

dest_id_hotel_cluster_count <- Trial_train[,sum_and_count(is_booking),by=list(orig_destination_distance, hotel_cluster)]
dest_id_hotel_cluster_count1 <- Trial_train[,sum_and_count(is_booking),by=list(srch_destination_id, hotel_cluster)]

dest_id_hotel_cluster_count #dest_id_hotel_cluster_count
dest_id_hotel_cluster_count1 #dest_id_hotel_cluster_count1 

top_five <- function(hc,v1){
  hc_sorted <- hc[order(v1,decreasing=TRUE)]
  n <- min(5,length(hc_sorted))
  paste(hc_sorted[1:n],collapse=" ")
}

dest_top_five <- dest_id_hotel_cluster_count[,top_five(hotel_cluster,V1),by=orig_destination_distance]
dest_top_five1 <- dest_id_hotel_cluster_count1[,top_five(hotel_cluster,V1),by=srch_destination_id]
dest_top_five  #dest_top_five 
dest_top_five1  #dest_top_five1 


Trial_test



