# This code is mainly based on versions of the "xgb-text2vec-tfidf" code i.e.,  posted by "danieleewww", "kailex", "konradb", "slickwilly", "tetyanayatsenko",  "tunguz"
# So many thanks!!!

library(tidyverse)
library(lubridate)
library(magrittr)
library(text2vec)
library(tokenizers)
library(stopwords)
library(Matrix)
library(stringr)
library(stringi)
library(forcats)
library(caret)
library(MLmetrics)
library(lightgbm) 
library(methods)
library(tictoc)

set.seed(325)

rm(list=ls())

tic("\n ... data loaded")
cat("\n Loading data ... \n \n")

train = read_csv("../input/train.csv", locale = locale(encoding = stringi::stri_enc_get()))
test = read_csv("../input/test.csv",locale = locale(encoding = stringi::stri_enc_get()))

toc()
tic("\n ... feature engineering completed")
cat("\n Feauture engineering ... \n")

tri <- 1:nrow(train)
y <- train$deal_probability

combined <- train %>% 
  select(-deal_probability) %>% 
  bind_rows(test) %>% 
   mutate(no_img = image %>% is.na() %>% as.integer(),
         no_dsc = description %>% is.na() %>% as.integer(),
         no_p1 = param_1 %>% is.na() %>% as.integer(), 
         no_p2 = param_2 %>% is.na() %>% as.integer(), 
         no_p3 = param_3 %>% is.na() %>% as.integer(),
         titl_len = str_length(title),
         desc_len = str_length(description),
         titl_capE = str_count(title, "[A-Z]"),
         titl_capR = str_count(title, "[А-Я]"),
         titl_lowE = str_count(title, "[a-z]"),
         titl_lowR = str_count(title, "[а-я]"),
         desc_cap = str_count(description, "[A-ZА-Я]"),
         titl_pun = str_count(title, "[[:punct:]]"),
         desc_pun = str_count(description, "[[:punct:]]"),
         titl_dig = str_count(title, "[[:digit:]]"),
         desc_dig = str_count(description, "[[:digit:]]"),
         user_type = as.factor(user_type),
         category_name = category_name %>% factor() %>% as.integer(),
         parent_category_name = parent_category_name %>% factor() %>% as.integer(), 
         region = region %>% factor() %>% as.integer(),
         param_1 = param_1 %>% factor() %>% as.integer(),
         param_2 = param_2 %>% factor() %>% as.integer(),
         param_3 = param_3 %>% factor() %>% fct_lump(prop = 0.00005) %>% as.integer(),
         city = city %>% factor() %>% fct_lump(prop = 0.0003) %>% as.integer(),
         user_id = user_id %>% factor() %>% fct_lump(prop = 0.000025) %>% as.integer(),
         price = log1p(price),
         txt = paste(title, description, sep = " "),
         mday = mday(activation_date),
         wday = wday(activation_date),         
         yday = yday(activation_date)) %>% 
         select(-item_id, -image, -title, -description, -activation_date) %>% 
         replace_na(list(image_top_1 = -1, price = -1, param_1 = -1, param_2 = -1, param_3 = -1, 
                          titl_cap = 0, desc_len = 0, desc_cap = 0, desc_pun = 0, 
                          desc_dig = 0, desc_capE = 0, desc_capR = 0)) %T>% 
         glimpse()

toc()
tic("\n ... text analysis completed")
cat("\n Text Analysis ... \n \n")
  
it <- combined %$%
str_to_lower(txt) %>%
str_replace_all("[^[:alpha:]]", " ") %>%
str_replace_all("\\s+", " ") %>%
tokenize_word_stems(language = "russian") %>% 
itoken()

vect <- create_vocabulary(it, ngram = c(1, 1), stopwords = stopwords("ru")) %>%
  prune_vocabulary(term_count_min = 3, doc_proportion_max = 0.4, vocab_term_max = 12500) %>% 
  vocab_vectorizer()

m_tfidf <- TfIdf$new(norm = "l2", sublinear_tf = T)

tfidf <-  create_dtm(it, vect) %>% 
  fit_transform(m_tfidf)

rm(it, vect, m_tfidf); gc()

toc()
cat("\n------------------ LGB model----------------------------\n")
tic("\n ... data prepared")
cat("\n Prepairing data ...\n")

X <- combined %>% 
  select(-txt) %>% 
  sparse.model.matrix(~ . - 1, .) %>% 
  cbind(tfidf)
 
rm(combined, tfidf); gc()

dtest <- X[-tri, ]
X <- X[tri, ]
tri <- caret::createDataPartition(y, p = 0.9, list = F) %>% c()
lgb.train = lgb.Dataset(data=X[tri, ], label=y[tri])
vals <- list(eval = lgb.train)

rm(X, y, tri); gc()

toc()
tic("\n ... model fitting completed")
cat("\n Fiting model ...\n")

print_round = 10
stop_round = 10
nroun =2000

lgb.params <- list(objective = "regression",
           boosting = "gbdt",
           learning_rate = 0.1,
           num_leaves = 31,
           num_threads = 8,
           max_depth = -1,
           min_child_samples = 20, 
           min_child_weight = 1e-3, 
           colsample_bytree = 0.7, 
           subsample = 0.8, 
           bagging_freq = 5,
           min_gain_to_split = 1e-4,
           max_bin = 255,
           min_data_in_bin=3,
           scale_pos_weight=95.5)

lgb.model <- lgb.train(params = lgb.params,
                       data = lgb.train,
                       eval = "rmse",
                       nrounds = nroun,
                       valids = vals,
                       eval_freq = print_round , 
                       early_stopping_rounds = stop_round)
  
lgb.importance(lgb.model, percentage = TRUE) %>%
    lgb.plot.importance(top_n = 35, measure = "Gain")

toc()
tic("\n ... prediction and submission completed")
cat("\n Apply method to test data and submit ...\n")

range01 <- function(x){(x-min(x))/(max(x)-min(x))}

prediction <- range01(predict(lgb.model, dtest))

read_csv("../input/sample_submission.csv") %>%  
mutate(deal_probability = prediction ) %>%
   write_csv(paste0("lgb_tfidf", lgb.model$best_score, ".csv"))

toc()

rm(list=ls())