# This R environment comes with all of CRAN preinstalled, as well as many other helpful packages
# The environment is defined by the kaggle/rstats docker image: https://github.com/kaggle/docker-rstats
# For example, here's several helpful packages to load in 

library(ggplot2) # Data visualization
library(readr) # CSV file I/O, e.g. the read_csv function

# Input data files are available in the "../input/" directory.
# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory

system("ls ../input")

# Any results you write to the current directory are saved as output.

#library(data.table)
#expedia_train <- fread('../input/train.csv', header=TRUE)
#expedia_test <- fread('../input/test.csv', header=TRUE)


#head(expedia_train)


#library(data.table)
#expedia_train <- fread('../input/train.csv', header=TRUE)
#expedia_test <- fread('../input/test.csv', header=TRUE)

#dest_id_hotel_cluster_count <- expedia_train[,length(is_booking),by=list(srch_destination_id, hotel_cluster)]

#top_five <- function(hc,v1){
#  hc_sorted <- hc[order(v1,decreasing=TRUE)]
#  n <- min(5,length(hc_sorted))
#  paste(hc_sorted[1:n],collapse=" ")
#}

#dest_top_five <- dest_id_hotel_cluster_count[,top_five(hotel_cluster,V1),by=srch_destination_id]

#dd <- merge(expedia_test,dest_top_five, by="srch_destination_id",all.x=TRUE)[order(id),list(id,V1)]

#setnames(dd,c("id","hotel_cluster"))

#write.csv(dd, file='submission_1.csv', row.names=FALSE)


#0	5 37 55 11 8
#1	5
#2	0 31 96 91 59
#3	1 45 79 24 54
#4	91 42 2 48 77
#5	91 42 16 48 33
#6	95 21 91 2 33
#7	95 91 18 68 98
#8	1 45 79 24 54
#9	55 32 10 80 50
#10	18 33 91 4 21
#11	25 38 6 75 82
#12	0 31 96 91 59



#a <- c(1,4)
#b <- c(2,1)
#c <- c(4,6)
#d <- c(4,3)
#e <- c(5,1)

#frame<-data.frame(a,b,c,d,e)
#frame

#tran <- t(frame)
#tran

#eu <- round(dist(tran),digits=2)
#eu

#{dist(tran)}^2

#den<-hclust(dist(tran)^2,method="single")
#plot(den)

#den <- hclust(dist(tran)^2,method="complete")
#den


library(data.table)
expedia_train <- fread('../input/train.csv', header=TRUE, select= c("is_booking","orig_destination_distance","hotel_cluster","srch_destination_id"))
expedia_test <- fread('../input/test.csv', header=TRUE)

sum_and_count <- function(x){
  sum(x)*0.8456 + length(x) *(1-0.8456)
}

dest_id_hotel_cluster_count <- expedia_train[,sum_and_count(is_booking),by=list(orig_destination_distance, hotel_cluster)]
dest_id_hotel_cluster_count1 <- expedia_train[,sum_and_count(is_booking),by=list(srch_destination_id, hotel_cluster)]


top_five <- function(hc,v1){
  hc_sorted <- hc[order(v1,decreasing=TRUE)]
  n <- min(5,length(hc_sorted))
  paste(hc_sorted[1:n],collapse=" ")
}

dest_top_five <- dest_id_hotel_cluster_count[,top_five(hotel_cluster,V1),by=orig_destination_distance]
dest_top_five1 <- dest_id_hotel_cluster_count1[,top_five(hotel_cluster,V1),by=srch_destination_id]

dd <- merge(expedia_test,dest_top_five, by="orig_destination_distance",all.x=TRUE)[order(id),list(id,V1)]

dd1 <- merge(expedia_test,dest_top_five1, by="srch_destination_id",all.x=TRUE)[order(id),list(id,V1)]

dd$V1[is.na(dd$V1)] <- dd1$V1[is.na(dd$V1)] 

setnames(dd,c("id","hotel_cluster"))

#write.csv(dd, file='submission_combo_merge.csv', row.names=FALSE)




#0	5 37 55 11 22
#1	5
#2	91
#3	1
#4	50 51
#5	91 42 28 95 48
#6	32 0
#7	95 91 18 98 68
#8	88
#9	55 32 10 34 50
#10	33 18 4 19 21
#11	64 46 82 62 36
#12	64 46 82 62 36



dest_id_hotel_cluster_count <- expedia_train[,length(is_booking),by=list(srch_destination_id, hotel_cluster)]

top_five <- function(hc,v1){
  hc_sorted <- hc[order(v1,decreasing=TRUE)]
  n <- min(5,length(hc_sorted))
  paste(hc_sorted[1:n],collapse=" ")
}

dest_top_five <- dest_id_hotel_cluster_count[,top_five(hotel_cluster,V1),by=srch_destination_id]

dd3 <- merge(expedia_test,dest_top_five, by="srch_destination_id",all.x=TRUE)[order(id),list(id,V1)]

#setnames(dd3,c("id","hotel_cluster"))

#write.csv(dd3, file='submission_1.csv', row.names=FALSE)


dd5 <- data.frame(dd,dd3)


dd6 <- cbind(dd5$id,dd5$hotel_cluster,dd5$V1)



colnames(dd6) = c("id", "hotel_cluster", "hotel_cluster2")

dd6


