library(tidyverse)
library(tidytext)
library(SnowballC)
library(openNLP)
library(topicmodels)
library(tm)
require(textstem)
library(wordcloud)
library(ggraph)
library(igraph)


train <- read_csv("../input/train.csv")
test <- read_csv("../input/test.csv")

head(train)

prop.table(table(train$target))

#Only 6% is Insincere content. 

Insincere_data <- train %>% filter(target == 1)

head(Insincere_data$question_text)

#lets analyze the Insincere data 

#unigram
t1 <- Insincere_data %>% unnest_tokens(word,question_text) %>% anti_join(stop_words,by="word")

t1 %>% count(word) %>% with(wordcloud(word,n,max.words = 50,color = c("purple4", "red4", "black")))

t2 <- t1 %>% mutate(word= lemmatize_words(word)) %>% count(word)
#word cloud
t2 %>% with(wordcloud(word,n,max.words = 50,color = c("purple4", "red4", "black")))


#bigram
t3 <- Insincere_data %>% unnest_tokens(bigram,question_text,token="ngrams",n=2)
t3_filt <- t3 %>% separate(bigram,c("word1","word2"),sep=" ") %>% 
  mutate(word1= lemmatize_words(word1), word2= lemmatize_words(word2)) %>%
  filter(!word1 %in% stop_words$word) %>% 
  filter(!word2 %in% stop_words$word)

t3_count <- t3_filt %>% count(word1,word2,sort=TRUE)

t3 <- t3_filt %>% unite(bigram, word1,word2,sep=" ") 
#word cloud
t3 %>% count(bigram)%>% with(wordcloud(bigram,n,max.words = 20,color = c("purple4", "red4", "black")))


#Network graph
bigram_graph <- t3_filt%>%count(word1, word2, sort = TRUE)%>%
  filter(n > 60)%>%graph_from_data_frame()

a <- grid::arrow(type = "closed", length = unit(.1, "inches"))

ggraph(bigram_graph)+
  geom_edge_link(aes(edge_alpha = n), show.legend = FALSE,arrow = a, end_cap = circle(.07, 'inches')) +
  geom_node_point(color = "lightblue", size = 3) +
  geom_node_text(aes(label = name), vjust = 1, hjust = 1) +
  theme_void()


#topic model

t_dtm <- t1%>% count(qid,word,sort=TRUE) %>% cast_dtm(qid,word,n)
t_dtm


t_lda <- LDA(t_dtm, k = 10, control = list(seed = 1234))
t_lda


t_topics <- tidy(t_lda,matrix="beta")
t_topics

top_terms <- t_topics %>% group_by(topic) %>% top_n(5,beta) %>% ungroup() %>% arrange(topic,-beta)
top_terms %>% mutate(term=reorder(term,beta)) %>% ggplot(aes(term,beta,fill=factor(topic)))+
  geom_col(show.legend = FALSE) +
  facet_wrap(~topic,scales = "free")+coord_flip()


t2_dtm <- t3 %>% count(qid,bigram,sort=TRUE) %>% cast_dtm(qid,bigram,n)
t2_lda <- LDA(t2_dtm, k = 5, control = list(seed = 1234))

t2_topics <- tidy(t2_lda,matrix="beta")
t2_topics

top2_terms <- t2_topics %>% group_by(topic) %>% top_n(5,beta) %>% ungroup() %>% arrange(topic,-beta)
top2_terms %>% mutate(term=reorder(term,beta)) %>% ggplot(aes(term,beta,fill=factor(topic)))+
  geom_col(show.legend = FALSE) +
  facet_wrap(~topic,scales = "free")+coord_flip()



#####Thanks
# https://www.kaggle.com/alaric81li215/eda-ml-for-beginners-by-a-beginner-bonus-on-qid
# https://www.tidytextmining.com/index.html
# https://www.kaggle.com/headsortails/treemap-house-of-horror-spooky-eda-lda-features
