#Modelling run 2. date 19/06/2018
#This is my second attempt for the Avito competition
#This uses the xgboost params and follows the form from https://www.kaggle.com/kailex/xgb-text2vec-tfidf-0-2237/code
#this run will follow the code format from kailex more closely as there is clearly merit to his method

rm(list=ls())

set.seed(18062018)
library(tidyverse)
library(forcats)
library(xgboost)
library(Matrix)
tr <- read_csv("../input/train.csv")
te <- read_csv("../input/test.csv")


tri <- 1:nrow(tr)
y <- tr$deal_probability

fct_lump_n <- 25
fct_lump_prop <- NULL
tr_te <- tr %>%
  select(-deal_probability) %>% 
  bind_rows(te) %>% 
  mutate(
    region = region %>% factor(exclude = NULL),
    city = city %>% factor(exclude = NULL) %>%  fct_lump(n=fct_lump_n, prop = fct_lump_prop),
    parent_category_name = parent_category_name %>% factor(exclude = NULL),
    category_name = category_name %>%  factor(exclude = NULL), #potentially lump,
    param_1 = param_1 %>% factor(exclude = NULL) %>% fct_lump(n=fct_lump_n, prop = fct_lump_prop),
    param_2 = param_2 %>% factor(exclude = NULL) %>% fct_lump(n=fct_lump_n, prop = fct_lump_prop),
    param_3 = param_3 %>% factor(exclude = NULL) %>% fct_lump(n=fct_lump_n, prop = fct_lump_prop),
    title = title %>% factor(exclude = NULL) %>% fct_lump(n=fct_lump_n, prop = fct_lump_prop),
    description = description %>% factor(exclude = NULL) %>% fct_lump(n=fct_lump_n, prop = fct_lump_prop),
    item_seq_number = item_seq_number %>% factor(exclude = NULL) %>% fct_lump(n=fct_lump_n, prop = fct_lump_prop),
    activation_date = activation_date %>% factor(exclude = NULL),
    user_type = user_type %>% factor(exclude = NULL),
    image_top_1 = image_top_1 %>% factor(exclude = NULL) %>% fct_lump(n=fct_lump_n, prop = fct_lump_prop)
  ) %>% 
  select(-c(item_id,user_id,image))
tr_te$price[is.na(tr_te$price)] <- median(tr$price,na.rm=T)

rm(tr, te); gc()

X <- tr_te %>% 
  sparse.model.matrix(~ . - 1, .)

rm(tr_te); gc()

dtest <- xgb.DMatrix(data = X[-tri, ])


X <- X[tri, ];
tri <- caret::createDataPartition(y, p = 0.9, list = F) %>% c()
dtrain <- xgb.DMatrix(data = X[tri, ], label = y[tri])
dval <- xgb.DMatrix(data = X[-tri, ], label = y[-tri])
cols <- colnames(X)

rm(X, y, tri); gc()


p <- list(objective = "reg:logistic",
          booster = "gbtree",
          eval_metric = "rmse",
          nthread = 8,
          eta = 0.05,
          max_depth = 18,
          min_child_weight = 11,
          gamma = 0,
          subsample = 0.8,
          colsample_bytree = 0.7,
          alpha = 2.25,
          lambda = 0,
          nrounds = 500)

m_xgb <- xgb.train(p, dtrain, p$nrounds, list(val = dval), print_every_n = 10, early_stopping_rounds = 5)
test <- read_rds("test.rds")
submission <- data.frame(item_id = test$item_id, deal_probability = predict(m_xgb,dtest))
write_csv(submission, "modelling_run2_submission_260167.csv")
