# Load libraries
pacman::p_load(jsonlite, tidyverse, tm, wordcloud)

# Import raw data
train <- fromJSON("../input/train.json")
train <- map_at(train, setdiff(names(train), c("photos", "features")), unlist) %>% tibble::as_tibble(.)

# Create TM corpus from unstructured text field
ds <- Corpus(VectorSource(train$description))

# Clean using tm_map
wordlist <- c("websiteredact",
              "kagglemanagerrenthopcombr")

ds <- tm_map(ds, tolower)
ds <- tm_map(ds, removeWords, wordlist)
ds <- tm_map(ds, removeWords, stopwords("english"))
ds <- tm_map(ds, stripWhitespace)
ds <- tm_map(ds, removePunctuation)
ds <- tm_map(ds, stemDocument)

# Create a 1-gram Term-Document Matrix
tdm.1g <- TermDocumentMatrix(ds) # 2- or 3-grams is also a good idea

# Have a look at most frequent terms
print("Terms appearing at least 5000 times:")
findFreqTerms(tdm.1g,  lowfreq = 5000)
print("Terms appearing at least 7000 times:")
findFreqTerms(tdm.1g,  lowfreq = 7000)

# Find correlates of some top terms
findAssocs(tdm.1g, "bathroom", .25)

# Remove Sparse Terms
tdm80.1g <- removeSparseTerms(tdm.1g, 0.8)

# Creates a Boolean matrix (counts # docs w/terms, not raw # terms)
tdm80.1g                <- inspect(tdm80.1g)
tdm80.1g[tdm80.1g >= 1] <- 1 
 
# Transform into a term-term adjacency matrix
termMatrix.1gram <- tdm80.1g %*% t(tdm80.1g)
 
# inspect subset of terms-term adj. mat.
termMatrix.1gram[1:10,1:10]
