# This R environment comes with all of CRAN preinstalled, as well as many other helpful packages
# The environment is defined by the kaggle/rstats docker image: https://github.com/kaggle/docker-rstats
# For example, here's several helpful packages to load in 

library(ggplot2) # Data visualization
library(readr) # CSV file I/O, e.g. the read_csv function
library(data.table)
library(sgd)
library(dplyr)

# Input data files are available in the "../input/" directory.
# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory

system("ls ../input")

# Any results you write to the current directory are saved as output.
clicks <- fread("../input/clicks_test.csv")

sortc <- clicks[order(clicks$ad_id),]

sortc$ad_id <- as.factor(sortc$ad_id)

aggregate <- as.data.frame(table(sortc$ad_id))

names(aggregate) <- c("ad_id", "count")

join <- merge(x = sortc, y = aggregate, by = "ad_id")

head(join)

join$display_id <- as.integer(join$display_id)

join$count <- as.integer(join$count)

join$ad_id <- as.integer(join$ad_id)

sorti <- join[order(join$display_id,-join$count),]

final <- sorti[!duplicated(sorti[, c("display_id")])]

head(final)

#final1 <- subset(final,select = c("display_id","ad_id")

#head(final1)



#write.csv(final1[order(-display_id),.(ad_id=paste(ad_id, collapse=" ")),by=display_id],"basic_submission.csv", row.names = FALSE)




