packages <- c("jsonlite", "dplyr", "purrr", "knitr", "stringr", "tidytext", "ggplot2", "grid", "tidyverse", "jpeg")
purrr::walk(packages, library, character.only = TRUE, warn.conflicts = FALSE)

data_dir <- "../input/"

# -----------------------------------------------------------------------------
# Load data and convert to tibble
# -----------------------------------------------------------------------------
train_org <- fromJSON(paste0(data_dir, "train.json"))
test <- fromJSON(paste0(data_dir, "test.json"))

# unlist every variable except `photos` and `features` and convert to tibble
vars <- setdiff(names(train_org), c("photos", "features"))

train_org <- map_at(train_org, vars, unlist) %>% tibble::as_tibble(.)
test <- map_at(test, vars, unlist) %>% tibble::as_tibble(.)

# Convert interest_level to factor
train_org$interest_level <- factor(train_org$interest_level, c("low", "medium", "high"))

# bind train and test data
test$interest_level <- NA
all <- rbind(train_org, test)

# -----------------------------------------------------------------------------
# FUNCTION to save and show images
# !!! THIS FUNCTION WORKS IN NORMAL LOCAL COMPUTING, BUT NOT WORK IN THIS KERNEL ENVIRONMENT.
# SO THE ALL SCRIPTS BELOW WHICH USE THIS FUNCTION IS COMMENTED-OUT.
# -----------------------------------------------------------------------------
saveShow_img <- function(data){
  photo_url <- lapply(1:length(data$photos), function(x) data$photos[[x]] %>% strsplit(., split=" ")) %>% unlist()
  print(paste0("reading and saving ", length(photo_url), " images"))
  sapply(1:length(photo_url), function(x) download.file(photo_url[x], paste0("data",x,".jpg"), mode = "wb"))
  print(paste0("showing ", length(photo_url), " images"))
  sapply(1:length(photo_url), function(x){
    data_img <- readJPEG(paste0("data", x, ".jpg"), native=TRUE)
    dev.new()
    plot(0:1, 0:1, type="n", ann=FALSE, axes=FALSE)
    rasterImage(data_img, 0, 0, 1, 1)})
}

# -----------------------------------------------------------------------------
# Basics: "photos"
# - number of photos
# -----------------------------------------------------------------------------
# number of photos by each listing
all$numOfPhotos <- sapply(1:nrow(all), function(x) length(unlist(all$photos[x])))
train_org$numOfPhotos <- sapply(1:nrow(train_org), function(x) length(unlist(train_org$photos[x])))


# Distribution of number of photos of TEST data is not very much different from that of TRAIN data
# Mode is 5 and most listings have 3-8 photos
# Maximum number of photos in TRAIN data is 68, in TEST data is 50
par(mfrow=c(3,1))
plot(table(all$numOfPhotos), xlim=c(0,70), ylim=c(0,20000), xlab="num of photos", ylab="num of listings", main="TRAIN+TEST: dist of num of photos")
plot(table(all[1:nrow(train_org),]$numOfPhotos), xlim=c(0,70), ylim=c(0,20000), xlab="num of photos", ylab="num of listings", main="TRAIN: dist of num of photos")
plot(table(all[1+nrow(train_org):nrow(all),]$numOfPhotos), xlim=c(0,70), ylim=c(0,20000), xlab="num of photos", ylab="num of listings", main="TEST: dist of num of photos")

# -----------------------------------------------------------------------------
# Basics: "photos"
# - Does number of photos impact to interest level ?
# -----------------------------------------------------------------------------
# It seems the listings with zero photos has high probability to be "LOW" interest_level
par(mfrow=c(3,1))
all[1:nrow(train_org),] %>% filter(interest_level=="high") %>% select(numOfPhotos) %>% unlist() %>% hist(., breaks=seq(0,70,by=1), xlim=c(0,70), ylim=c(0,6000), xlab="num of photos", ylab="num of listings", main="HIGH listings")
all[1:nrow(train_org),] %>% filter(interest_level=="medium") %>% select(numOfPhotos) %>% unlist() %>% hist(., breaks=seq(0,70,by=1), xlim=c(0,70), ylim=c(0,6000), xlab="num of photos", ylab="num of listings", main="MEDIUM listings")
all[1:nrow(train_org),] %>% filter(interest_level=="low") %>% select(numOfPhotos) %>% unlist() %>% hist(., breaks=seq(0,70,by=1), xlim=c(0,70), ylim=c(0,6000), xlab="num of photos", ylab="num of listings", main="LOW listings")

# table of number of photos and number of listings by interest_level
# numOfPhotos_tbl <- spread(data.frame(xtabs(data=all[1:nrow(train_org),], ~ interest_level + numOfPhotos)), key=interest_level, val=Freq) %>% arrange(numOfPhotos)
numOfPhotos_tbl <- spread(data.frame(xtabs(data=train_org, ~ interest_level + numOfPhotos)), key=interest_level, val=Freq) %>% arrange(numOfPhotos)
numOfPhotos_tbl <- numOfPhotos_tbl %>%
  mutate(sum=low+medium+high) %>% mutate(lowRatio=low/sum, medRatio=medium/sum, highRatio=high/sum) %>%
  mutate(medHigh=medium+high, medHighRatio=medHigh/sum)
numOfPhotos_tbl

# if number of photos = 0, high probability of "LOW" interest_level
# if number of photos = 1 - 15, the ratio of "Medium" or "High" interest_level is more than 20%
# if number of photos >= 16, it seems that ratio of "LOW" interest_level is increased.
par(mfrow=c(1,1))
plot(numOfPhotos_tbl$lowRatio, type="b", ylim=c(0, 1.0), lty=2, col="red", xlab="num of photos", ylab="ratio of listings", main="ratio of listings by num of photos", xaxt="n")
par(new=T); plot(numOfPhotos_tbl$medRatio, type="b", ylim=c(0, 1.0), lty=2, col="blue", xlab="", ylab="", xaxt="n", yaxt="n");
par(new=T); plot(numOfPhotos_tbl$highRatio, type="b", ylim=c(0, 1.0), lty=2, lwd=2, col="black", xlab="", ylab="", xaxt="n", yaxt="n");
par(new=T); plot(numOfPhotos_tbl$medHighRatio, type="b", ylim=c(0, 1.0), lty=1, lwd=3, col="black", xlab="", ylab="", xaxt="n", yaxt="n");

# -----------------------------------------------------------------------------
# Basics: "photos"
# - Check photos of the listing with many photos >= 16
# -----------------------------------------------------------------------------
# Show the data of number of photos = 22
# Can see some same managers, so sort by manager_id
var <- c("numOfPhotos", "manager_id", "building_id", "created", "interest_level", "photos")
tmp_ <- all %>% filter(numOfPhotos==22) %>% arrange(manager_id)
kable(tmp_[,var]) %>% head(20)

# Check the photos of some managers...
# building_id and manager_is id different, but It seems those are almost same set of photos !!??

# saveShow_img(data=tmp_[7,])
# saveShow_img(data=tmp_[8,])
# saveShow_img(data=tmp_[15,])

# graphics.off()


# -----------------------------------------------------------------------------
# Basics: "bathrooms"
# - number of bathrooms
# -----------------------------------------------------------------------------
# Most of listings have 1 bathroom
# Maximum number of bathrooms in TRAIN data is 10, but in TEST data is 112 !!!
table(all$bathrooms)

summary(all$bathrooms)
summary(train_org$bathrooms)
summary(test$bathrooms)

var <- c("bathrooms", "bedrooms", "numOfPhotos", "price", "interest_level", "features")

# Show the data of bathrooms = 112
# Checked the photos of this listings, but it does not seem that this have so many bathrooms !! or so large bathroom !!
# Price is also relatively cheap
tmp_ <- all %>% filter(bathrooms==112)
kable(tmp_[,var])
# saveShow_img(data=tmp_)
# graphics.off()

# How about the data of bathrooms = 20 ?
tmp_ <- all %>% filter(bathrooms==20)
kable(tmp_[,var])
# saveShow_img(data=tmp_[1,])
# graphics.off()

# saveShow_img(data=tmp_[2,])
# graphics.off()

# How about the data of bathrooms = 10 ?
tmp_ <- all %>% filter(bathrooms==10)
kable(tmp_[,var])
# saveShow_img(data=tmp_)
# graphics.off()

# How about the data of bathrooms = 7.5
tmp_ <- all %>% filter(bathrooms==7.5)
kable(tmp_[,var])
# saveShow_img(tmp_)
# graphics.off()

# -----------------------------------------------------------------------------
# Basics: "bathrooms"
# - Does "bathrooms" impact to interest level ?
# -----------------------------------------------------------------------------
# bathtooms and number of listings by interest_level
# bathrooms_tbl <- spread(data.frame(xtabs(data=all[1:nrow(train_org),], ~ bathrooms + interest_level)), key=interest_level, val=Freq)
bathrooms_tbl <- spread(data.frame(xtabs(data=train_org, ~ bathrooms + interest_level)), key=interest_level, val=Freq)
bathrooms_tbl <- bathrooms_tbl %>%
  mutate(sum=low+medium+high) %>% mutate(lowRatio=low/sum, medRatio=medium/sum, highRatio=high/sum) %>%
  mutate(medHigh=medium+high, medHighRatio=medHigh/sum)
bathrooms_tbl

# if number of bathrooms = 0, high probability of "LOW" interest_level
# if number of bathrooms >= 2.5, it seems that high ratio of "LOW" interest_level....
par(mfrow=c(1,1))
plot(bathrooms_tbl$lowRatio, type="b", ylim=c(0, 1.0), lty=2, col="red", xlab="bathrooms", ylab="ratio of listings", main="ratio of listings by bathrooms", xaxt="n")
par(new=T); plot(bathrooms_tbl$medRatio, type="b", ylim=c(0, 1.0), lty=2, col="blue", xlab="", ylab="", xaxt="n", yaxt="n");
par(new=T); plot(bathrooms_tbl$highRatio, type="b", ylim=c(0, 1.0), lty=2, lwd=2, col="black", xlab="", ylab="", xaxt="n", yaxt="n");
par(new=T); plot(bathrooms_tbl$medHighRatio, type="b", ylim=c(0, 1.0), lty=1, lwd=3, col="black", xlab="", ylab="", xaxt="n", yaxt="n");

# -----------------------------------------------------------------------------
# Basics: "bedrooms"
# - number of bedrooms
# -----------------------------------------------------------------------------
# Most of listings have 0-3 bedrooms
# Maximum number of bathrooms in TRAIN data is 8, and in TEST data is 7
table(all$bedrooms)

summary(all$bedrooms)
summary(train_org$bedrooms)
summary(test$bedrooms)

var <- c("bathrooms", "bedrooms", "numOfPhotos", "price", "interest_level", "features")

# Show the data of bedrooms = 8
# first one has no photo, but second one has.
# Second one is moderately expensive: price = 9995
tmp_ <- all %>% filter(bedrooms==8)
kable(tmp_[,var])
# saveShow_img(data=tmp_[2,])
# graphics.off()

# How about the data of bedrooms = 7 ?
# forth one has bathrooms = 1.0, but others have more than 2.5
# let us check fourth one and fifth one
tmp_ <- all %>% filter(bedrooms==7)
kable(tmp_[,var])
# saveShow_img(data=tmp_[4,])
# graphics.off()

# saveShow_img(data=tmp_[5,])
# graphics.off()

# How about the data of bedrooms = 0 ?
# Can see some same manager, so sort by manager_id
# fourth one: really bedrooms = 0 ?
tmp_ <- all %>% filter(bedrooms==0) %>% arrange(manager_id)
kable(tmp_[1:30,var])

# saveShow_img(data=tmp_[4,])
# graphics.off()

# -----------------------------------------------------------------------------
# Basics: "bedrooms"
# - Does "bedrooms" impact to interest level ?
# -----------------------------------------------------------------------------
# bathtooms and number of listings by interest_level
# bedrooms_tbl <- spread(data.frame(xtabs(data=all[1:nrow(train_org),], ~ bedrooms + interest_level)), key=interest_level, val=Freq)
bedrooms_tbl <- spread(data.frame(xtabs(data=train_org, ~ bedrooms + interest_level)), key=interest_level, val=Freq)
bedrooms_tbl <- bedrooms_tbl %>%
  mutate(sum=low+medium+high) %>% mutate(lowRatio=low/sum, medRatio=medium/sum, highRatio=high/sum) %>%
  mutate(medHigh=medium+high, medHighRatio=medHigh/sum)
bedrooms_tbl

# if number of bedrooms >= 6, it seems that high ratio of "LOW" interest_level....
par(mfrow=c(1,1))
plot(bedrooms_tbl$lowRatio, type="b", ylim=c(0, 1.0), lty=2, col="red", xlab="bedrooms", ylab="ratio of listings", main="ratio of listings by bedrooms", xaxt="n")
par(new=T); plot(bedrooms_tbl$medRatio, type="b", ylim=c(0, 1.0), lty=2, col="blue", xlab="", ylab="", xaxt="n", yaxt="n");
par(new=T); plot(bedrooms_tbl$highRatio, type="b", ylim=c(0, 1.0), lty=2, lwd=2, col="black", xlab="", ylab="", xaxt="n", yaxt="n");
par(new=T); plot(bedrooms_tbl$medHighRatio, type="b", ylim=c(0, 1.0), lty=1, lwd=3, col="black", xlab="", ylab="", xaxt="n", yaxt="n");

# -----------------------------------------------------------------------------
# Basics: "price"
# - Distribution of price and range
# -----------------------------------------------------------------------------
# Price has large range but Median is same in TRAIN and TEST data = 3150, Mean is also close.
# Minimum price in TEST data is 1, maximum price in TRAIN is 4490000 !!!
summary(all$price)
summary(train_org$price)
summary(test$price)

# select variables to show
var <- c("bathrooms", "bedrooms", "numOfPhotos", "price", "interest_level", "features", "description")

# Show the data of price = 4490000
# price is very expensive, but bedrooms = 2 and bathrooms = 1, and no photos
tmp_ <- all %>% filter(price==4490000)
kable(tmp_[,var])

# How about the data of price = 1 ?
# only 1 record, almost no information.
# photos says "seal" and photos URL is "http://sealagreements.com/leasing/"
# It seems that price information is just entered for some reasons
tmp_ <- all %>% filter(price==1)
kable(tmp_[,var])
# saveShow_img(data=tmp_[1,])
# graphics.off()

# Check the price is price <= 100
# 3 records including price = 1
# Check the second one. This seems to be some (sharing) offices.
tmp_ <- all %>% filter(price<=100)
kable(tmp_[,var])
# saveShow_img(data=tmp_[2,])
# graphics.off()

# Check the price is price >= 100,000
# Many of them do not have photos
# Check the tenth one. This seems to be expepensive mansion.
var <- c("bathrooms", "bedrooms", "numOfPhotos", "price", "interest_level", "features")
tmp_ <- all %>% filter(price>=100000)
nrow(tmp_)
kable(tmp_[,var])
# saveShow_img(data=tmp_[10,])
# graphics.off()

# -----------------------------------------------------------------------------
# Basics: "price"
# - Does "price" impact to interest level ?
# -----------------------------------------------------------------------------
# Discretize price
# all$priceCat <- cut(all$price, breaks=c(0, 2^(5:23)))
train_org$priceCat <- cut(train_org$price, breaks=c(0, 2^(5:23)))

# bathtooms and number of listings by interest_level
# priceCat_tbl <- spread(data.frame(xtabs(data=all[1:nrow(train_org),], ~ priceCat + interest_level)), key=interest_level, val=Freq)
priceCat_tbl <- spread(data.frame(xtabs(data=train_org, ~ priceCat + interest_level)), key=interest_level, val=Freq)
priceCat_tbl <- priceCat_tbl %>%
  mutate(sum=low+medium+high) %>% mutate(lowRatio=low/sum, medRatio=medium/sum, highRatio=high/sum) %>%
  mutate(medHigh=medium+high, medHighRatio=medHigh/sum)
priceCat_tbl

# if price between 512 and 2048 have high probability to be "High" interest_level
# if price betwenn 2048 and 8192 have high probability to be "Medium" interest_level
par(mfrow=c(1,1))
plot(priceCat_tbl$lowRatio, type="b", ylim=c(0, 1.0), lty=2, col="red", xlab="priceCat", ylab="ratio of listings", main="ratio of listings by priceCat", xaxt="n")
par(new=T); plot(priceCat_tbl$medRatio, type="b", ylim=c(0, 1.0), lty=2, col="blue", xlab="", ylab="", xaxt="n", yaxt="n");
par(new=T); plot(priceCat_tbl$highRatio, type="b", ylim=c(0, 1.0), lty=2, lwd=2, col="black", xlab="", ylab="", xaxt="n", yaxt="n");
par(new=T); plot(priceCat_tbl$medHighRatio, type="b", ylim=c(0, 1.0), lty=1, lwd=3, col="black", xlab="", ylab="", xaxt="n", yaxt="n");
