# Following the example https://quanteda.io/articles/pkgdown/replication/text2vec.html
library(text2vec)
library(quanteda)

# Build document-term matrix first
dtr <- read.csv("../input/train.csv", stringsAsFactors=FALSE)
dts <- read.csv("../input/test.csv", stringsAsFactors=FALSE)
dts$target <- NA
d <- rbind(dtr,dts)
rm(dtr,dts)

tokens(tolower(d$question_text), remove_numbers = TRUE, remove_punct = TRUE,
  remove_symbols = TRUE, remove_separators = TRUE,
  remove_twitter = TRUE, remove_hyphens = TRUE, remove_url = TRUE) %>%
  tokens_remove(stopwords("english")) %>%
  tokens_wordstem -> qtokens 
  
q_dtm <- dfm(qtokens)

# uses features from a trimmed DTM to only reasonably frequent terms
qfeats <- dfm_trim(q_dtm, min_count = 5) %>% featnames()

# trim the tokens set
# leave the pads so that non-adjacent words will not become adjacent
qtoks <- tokens_select(qtokens, qfeats, padding = TRUE)

# Construct the feature co-occurrence matrix
qfcm <- fcm(qtoks, context = "window", count = "weighted", weights = 1 / (1:5), tri = TRUE)

# Learn GloVe embedding vectors
glove <- GlobalVectors$new(word_vectors_size = 50, vocabulary = featnames(qfcm), x_max = 10)
# Main vectors
q_main_vecs_50 <- fit_transform(qfcm, glove, n_iter = 1000)
# Use context vectors as well
q_context_vecs_50 <- t(glove$components)

# Save the results
save(q_main_vecs_50, q_context_vecs_50, qtoks, qfeats, file="learned_glove_vectors.Rda")