#This is a script implementing a simple genetic algorithm for frequency, time lags and target encoding based features automatic engineering and evaluation accross the whole dataset.
#This weekend I had some time to try improving my previous script (https://www.kaggle.com/alesgb/genetic-feature-engineering-and-evaluation/) to find better features. 
#I hope the script I made could be benefitial for the participants to improve the scores.


#The script selects datas = N (currently set to just 5 for time related reasons) random samples from the entire dataset uniformly and for each of the datasets performs giter = M (currently set to just 10 for time related reasons) genetic populations of various frequency based features
#The selection criteria for the features to make it to the next population is being present in the set of most important features provided by lgbm. Statistics for the overall and mean importance 
#of all of the features is saved in hash table feathash() and is based upon the sum of (fi$Gain[i]+fi$Cover[i] + fi$Frequency[i])*maxauc/3 accross all runs for all datasets (giter*datas). Also maximal AUC foond per feature being
#in the model is reported.
#Here maxauc is the AUC on the current validation data set for the current run. The features with the highest rankings (hopefully are homogeniously important for the whole data) and can be used on the entire dataset in your final analysis. 

#The target encoding is done in the cross validated fashion to aviod overfitting. 

#Final feature importance based on the performed set of runs is saved in 0finalfeatimportance.csv!!!

#Comparing to the previous script I:

#1) Added time lagged features and target cncodingss in the feature engineering part.
#2) Added a normalized quality of the feature measure.
#3) Separated the validation and training parts by days.


#p.s. This script was supposed to generate optimal sets and scores of features, but in practice it didn't work perfectly
#in the sense that the top 20 features applied together would not guarantee optimal scores on the larger scale cross validation 
#or leaderboard
#would be interested to see the final sets of winning features after the end of the competition 
#and then figure out how and if they could have been found by this approach as well as why they haven't been found significant

#do not hesitate to ask questions and leave comments if you have any.
#many thanks to the guys from the older puclic scripts, from whom I borrowed parts of the code. 

#GOOD LUCK AND HAVE FUN! :) 

library(dplyr)
library(data.table)
library(magrittr)
library(knitr)
library(tidyverse)
library(lubridate)
library(zoo)
library(DescTools)
library(lightgbm)
library(hash)
#######################################################
# Some frequently used control parameters:
####################################################### 
testing = T
#######################################################
testing_size =  500000
#######################################################
val_ratio = 0.90
#######################################################
####  fit parts of data seprately to save memory  #####
partition = 1   
psize = 35
train_path = "../input/talkingdata-adtracking-fraud-detection/train.csv"
test_path  = "../input/talkingdata-adtracking-fraud-detection/test.csv"
fw_train_deltas_path = "../input/td-forward-time-deltas-as-csv/td_forward_time_deltas.csv"
bw_train_deltas_path = "../input/td-backward-time-deltas-as-csv/td_time_deltas.csv"
fw_test_deltas_path = "../input/td-test-forward-time-deltas-as-csv/td_test_forward_time_deltas.csv"
bw_test_deltas_path = "../input/td-test-backward-time-deltas-as-csv/td_test_time_deltas.csv"

#######################################

train_col_names = c("ip", "app", "device", "os", "channel", 
                    "click_time", "attributed_time", "is_attributed")
test_col_names = c("click_id","ip","app","device","os","channel","click_time")

#######################################################

most_freq_hours_in_test_data = c("4","5","9","10","13","14")
least_freq_hours_in_test_data = c("6","11","15")

#######################################################

total_rows = 184903890
# A function for processing the train/test data

#######################################################

process = function(df) {
cat("Building new features...\n")
df = df %>% mutate(wday = Weekday(click_time), 
                   hour = Hour(click_time),
                   minute = Minute(click_time),
                   second = Second(click_time),
                   time = wday*24*60*60 + hour*60*60+minute*60+second,
                   intesthh = ifelse(hour %in% most_freq_hours_in_test_data, 1,
                                       ifelse(hour %in% least_freq_hours_in_test_data, 3, 2))) 
return(as.data.table(df))
}
  
#######################################################
rstart = 22558000
rstartvalid = 82282000
rstartvalid2 = 144687300
datas = 75 #set to 5 for being quickly feasible on Kaggle server
giter = 10 #set to 10 for being quickly feasible on Kaggle server
run=1
feathash = hash()
res = as.data.frame(array(0,dim = c(18790469,3)))
for(d in runif(5,1,datas))
{
  
  bestauc = 0 
  maxauc = 0
  cat("Begin Processing Dataset: ", d, " rows. \n")
  train_rows = testing_size
  skip_rows_train =  testing_size*(d-1)+rstart
  skip_rows_valid =  as.integer(testing_size*(d-0.5)+rstartvalid)
  skip_rows_valid2 =  as.integer(testing_size*(d-0.5)+rstartvalid2)
  test_rows = testing_size*0.1
  skip_rows_test = test_rows*(d-1)
  
  #######################################################
  
  #*****************************************************************

  #######################################################
  
  #*****************************************************************
  # Preparing the training data
  
  #######################################################
  
  
  #skip_rows_train = 22558275+74.4*500000#184903890/4*0.488
  #train_rows =10
  cat("Reading the target encoding data...\n")
  train = fread(train_path, skip = skip_rows_train, nrows = train_rows, colClasses = list(numeric=1:5),
                showProgress = FALSE, col.names = train_col_names) %>% select(-c(attributed_time))
  fw_deltas = fread(fw_train_deltas_path, skip = skip_rows_train, nrows = train_rows, 
                    colClasses = "numeric", showProgress = FALSE, col.names = "forward_time_delta")
  train = cbind( train, fw_deltas )
  rm(fw_deltas)
  invisible(gc())
  #head(train)
  
    
  cat("Reading the training/validation data...\n")
  valid = fread(train_path, skip = skip_rows_valid, nrows = train_rows, colClasses = list(numeric=1:5),
                showProgress = FALSE, col.names = train_col_names) %>% select(-c(attributed_time))
  fw_deltas = fread(fw_train_deltas_path, skip = skip_rows_valid, nrows = train_rows, 
                    colClasses = "numeric", showProgress = FALSE, col.names = "forward_time_delta")
  valid = cbind(valid, fw_deltas )
  rm(fw_deltas)
  invisible(gc())
  #head(train)
  
   cat("Reading the training/validation data...\n")
  valid2 = fread(train_path, skip = skip_rows_valid2, nrows = train_rows, colClasses = list(numeric=1:5),
                showProgress = FALSE, col.names = train_col_names) %>% select(-c(attributed_time))
  fw_deltas = fread(fw_train_deltas_path, skip = skip_rows_valid2, nrows = train_rows, 
                    colClasses = "numeric", showProgress = FALSE, col.names = "forward_time_delta")
  valid2 = cbind(valid2, fw_deltas )
  rm(fw_deltas)
  invisible(gc())
  #head(train)

  
  cat("Reading the test data: ", test_rows, " rows. \n")
  test = fread(test_path, nrows=test_rows,skip = skip_rows_test,col.names = test_col_names, colClasses=list(numeric=2:6), showProgress = FALSE)
  fw_deltas = fread(fw_test_deltas_path, nrows = test_rows,skip = skip_rows_test, colClasses = "numeric", showProgress = FALSE)
  test = cbind( test, fw_deltas[,2])
  names(test)[dim(test)[2]]="forward_time_delta"
  rm(fw_deltas)
  invisible(gc())
  
  
  train$order=1:train_rows
  train$valid = 0
  valid$valid = 1
  valid2$valid = 3
  test$valid = 2
  valid$order =1:train_rows
  valid$click_id = -1
  isatrval2 = valid2$is_attributed
  valid2$order =1:train_rows
  valid2$click_id = -1
  isatrval = valid$is_attributed
  test$order = 1:test_rows
  train$click_id = -1
 
 

  #head(valid$forward_time_delta)
  
  valid$is_attributed = -1#valid$is_attributed + 5
  valid2$is_attributed = -1
  test$is_attributed = -1 
  train = merge(train, test, all=T)
  train = merge(train, valid, all=T)
  train = merge(train, valid2, all=T)
  #rm(test)
  #rm(valid)
  invisible(gc())
  #######################################################
  
  
  
  #######################################################
  
  train = process(train)
  
  lkeep = length(train)-1
  
  g = 0
 
  
  tnames = names(train)[-c(6,8,9,10,11,12,14,15)]
  
  if(run ==1)
  {
    run = 2
    
    for(nm in names(train))
      feathash[[nm]]= c(as.numeric(0),as.numeric(1),as.numeric(0))
  }  
  
  i=1
  cat("Begin feature engineering \n")
  while(i<=psize)
  {
    #print(i)
    a = list()
    intorder = rbinom(n = 1,size = 10,prob = 0.1)+2
    nc=NULL
    for(j in 1:intorder)
    {
      cura = tnames[sample.int(size = 1,n = length(tnames))]
      nc =c(nc,cura)
      a[[cura]]=train[[cura]]
    }

    fnamec = paste0("count_",paste0(sort(unique(nc)),collapse ="_"))
    if(!fnamec %in% names(train))
        invisible(train[, paste0(fnamec):=.N, by=a])
    if(max(train[[fnamec]])==1)
    {
      train[[fnamec]] = NULL
      i=i-1
    }else
    {
      if(!has.key(hash = feathash,key =  fnamec))
        feathash[[fnamec]] = c(as.numeric(0),as.numeric(1),as.numeric(0))
      
      fnameo = paste0("occur_",paste0(sort(unique(nc)),collapse ="_"))
      
      if(!has.key(hash = feathash,key =  fnameo))
        feathash[[fnameo]] = c(as.numeric(0),as.numeric(1),as.numeric(0))
      
      if(!fnameo %in% names(train) && runif(1,0,1)<0.1)
        invisible(train[, paste0(fnameo):=1:.N, by=a])
     
      fnameo = paste0("sdhour_",paste0(sort(unique(nc)),collapse ="_"))
      
      if(!has.key(hash = feathash,key =  fnameo))
        feathash[[fnameo]] = c(as.numeric(0),as.numeric(1),as.numeric(0))
      
      if(!fnameo %in% names(train) && runif(1,0,1)<0.9)
        invisible(train[, paste0(fnameo):=sd(hour), by=a])    
        
      fnameo = paste0("meanhour_",paste0(sort(unique(nc)),collapse ="_"))
      
      if(!has.key(hash = feathash,key =  fnameo))
        feathash[[fnameo]] = c(as.numeric(0),as.numeric(1),as.numeric(0))
      
      if(!fnameo %in% names(train) && runif(1,0,1)<0.9)
        invisible(train[, paste0(fnameo):=mean(hour), by=a])     
        
       fnameo = paste0("sdattrib_",paste0(sort(unique(nc)),collapse ="_"))
      
      if(!has.key(hash = feathash,key =  fnameo))
        feathash[[fnameo]] = c(as.numeric(0),as.numeric(1),as.numeric(0))
      
      if(!fnameo %in% names(train) && runif(1,0,1)<0.9)
        invisible(train[, paste0(fnameo):=sd(is_attributed), by=a])    
        
      fnameo = paste0("meanattrib_",paste0(sort(unique(nc)),collapse ="_"))
      
      if(!has.key(hash = feathash,key =  fnameo))
        feathash[[fnameo]] = c(as.numeric(0),as.numeric(1),as.numeric(0))
      
      if(!fnameo %in% names(train) && runif(1,0,1)<0.9)
        invisible(train[, paste0(fnameo):=mean(is_attributed), by=a])     
        
        fnameo = paste0("cumattrib_",paste0(sort(unique(nc)),collapse ="_"))
      
      if(!has.key(hash = feathash,key =  fnameo))
        feathash[[fnameo]] = c(as.numeric(0),as.numeric(1),as.numeric(0))
      
      if(!fnameo %in% names(train) && runif(1,0,1)<0.9)
        invisible(train[, paste0(fnameo):=cumsum(is_attributed), by=a])     
           
        
        if(runif(1,0,1)<0.5)
        { 
            fcura = paste0("lag_appf",paste0(sort(unique(nc)),collapse ="_"))
            
            if(!fcura %in% names(train) && runif(1,0,1)<0.2)
                invisible(train[, paste0(fcura) := c(-9999, app[-length(app)]), by = a])
            if(!has.key(hash = feathash,key = fcura))
            feathash[[fcura]] = c(as.numeric(0),as.numeric(1),as.numeric(0))
        }else{
            
            fcura = paste0("lag_appb",paste0(sort(unique(nc)),collapse ="_"))
             if(!fcura %in% names(train) && runif(1,0,1)<0.2)
            invisible(train[, paste0(fcura) :=c(app[-1],-9999), by = a])
            if(!has.key(hash = feathash,key = fcura))
            feathash[[fcura]] = c(as.numeric(0),as.numeric(1),as.numeric(0))
        }
        
      if(runif(1,0,1)<0.5)
      { 
        fcura = paste0("lag_ipf",paste0(sort(unique(nc)),collapse ="_"))
        if(!fcura %in% names(train) && runif(1,0,1)<0.2)
          invisible(train[, paste0(fcura) := c(-9999, ip[-length(ip)]), by = a])
        if(!has.key(hash = feathash,key = fcura))
          feathash[[fcura]] = c(as.numeric(0),as.numeric(1),as.numeric(0))
      }else{
        fcura = paste0("lag_ipb",paste0(sort(unique(nc)),collapse ="_"))
        if(!fcura %in% names(train) && runif(1,0,1)<0.2)
          invisible(train[, paste0(fcura) :=c(ip[-1],-9999), by = a])
        if(!has.key(hash = feathash,key = fcura))
          feathash[[fcura]] = c(as.numeric(0),as.numeric(1),as.numeric(0))
      }
        
        if(runif(1,0,1)<0.5)
        { 
            fcura = paste0("lag_chf",paste0(sort(unique(nc)),collapse ="_"))
             if(!fcura %in% names(train) && runif(1,0,1)<0.2)
            invisible(train[, paste0(fcura) := c(-9999, channel[-length(channel)]), by = a])
            if(!has.key(hash = feathash,key = fcura))
            feathash[[fcura]] = c(as.numeric(0),as.numeric(1),as.numeric(0))
        }else{
            fcura = paste0("lag_chb",paste0(sort(unique(nc)),collapse ="_"))
            invisible(train[, paste0(fcura) :=c(channel[-1],-9999), by = a])
            if(!has.key(hash = feathash,key = fcura))
            feathash[[fcura]] = c(as.numeric(0),as.numeric(1),as.numeric(0))
        }
        
        if(runif(1,0,1)<0.5)
        { 
            fcura = paste0("lag_osf",paste0(sort(unique(nc)),collapse ="_"))
             if(!fcura %in% names(train) && runif(1,0,1)<0.2)
            invisible(train[, paste0(fcura) := c(-9999,os[-length(os)]), by = a])
            if(!has.key(hash = feathash,key = fcura))
            feathash[[fcura]] = c(as.numeric(0),as.numeric(1),as.numeric(0))
        }else{
            fcura = paste0("lag_osb",paste0(sort(unique(nc)),collapse ="_"))
             if(!fcura %in% names(train) && runif(1,0,1)<0.2)
            invisible(train[, paste0(fcura) :=c(os[-1],-9999), by = a])
        if(!has.key(hash = feathash,key = fcura))
            feathash[[fcura]] = c(as.numeric(0),as.numeric(1),as.numeric(0))
        }
        
        if(runif(1,0,1)<0.5)
        { 
            fcura = paste0("lag_devf",paste0(sort(unique(nc)),collapse ="_"))
             if(!fcura %in% names(train) && runif(1,0,1)<0.2)
            invisible(train[, paste0(fcura) := c(-9999,device[-length(device)]), by = a])
            if(!has.key(hash = feathash,key = fcura))
            feathash[[fcura]] = c(as.numeric(0),as.numeric(1),as.numeric(0))
        }else{
            fcura = paste0("lag_devb",paste0(sort(unique(nc)),collapse ="_"))
             if(!fcura %in% names(train) && runif(1,0,1)<0.2)
            invisible(train[, paste0(fcura) :=c(device[-1],-9999), by = a])
            if(!has.key(hash = feathash,key = fcura))
            feathash[[fcura]] = c(as.numeric(0),as.numeric(1),as.numeric(0))
        }
        
        if(runif(1,0,1)<0.5)
        { 
            fcura = paste0("lag_timef",paste0(sort(unique(nc)),collapse ="_"))
             if(!fcura %in% names(train) && runif(1,0,1)<0.2)
            invisible(train[, paste0(fcura) := c(-9999,time[-length(time)]), by = a])
            if(!has.key(hash = feathash,key = fcura))
            feathash[[fcura]] = c(as.numeric(0),as.numeric(1),as.numeric(0))
        }else{
            fcura = paste0("lag_timeb",paste0(sort(unique(nc)),collapse ="_"))
             if(!fcura %in% names(train) && runif(1,0,1)<0.2)
            invisible(train[, paste0(fcura) :=c(time[-1],-9999), by = a])
        if(!has.key(hash = feathash,key = fcura))
            feathash[[fcura]] = c(as.numeric(0),as.numeric(1),as.numeric(0))
        }
        
         
    }
    i = i+1
    
  }
  
  
  for(g in 1:giter)
  {
    invisible(gc())
    cat("Begin iteration: ", i, " rows for dataset.",d," \n")
    #######################################################
    
    cat("The training set has", nrow(train), "rows and", ncol(train), "columns.\n")
    #cat("The column names of the train are: \n")
    #cat(colnames(train), "\n")
    print("The size of the train is: ") 
    print(object.size(train), units = "auto")
    
    #######################################################
    
    #*****************************************************************
    # Modelling
    
    #######################################################
    
    print("The table of class unbalance")
    table(train$is_attributed)
    
    #train$ip = NULL
    #test = train[train$is_attributed==-1,]
    dtrain = train[train$is_attributed ==-1 & train$valid ==3 ,]
    valid  = train[train$is_attributed ==-1 & train$valid ==1 ,]
    test = train[train$is_attributed ==-1 & train$valid ==2 ,]
    dtrain  =  dtrain[order(order)]
    valid =  valid[order(order)]
    test = test[order(order)]
    valid$order = NULL
    dtrain$order = NULL
    test$order = NULL
    dtrain$click_time = NULL
    valid$click_time = NULL
    test$click_time = NULL
    dtrain$click_id = NULL
    valid$click_id = NULL
    valid$valid = NULL
    test$valid = NULL
    dtrain$valid = NULL
    valid$is_attributed = isatrval 
    dtrain$is_attributed = isatrval2 
    
    #dtrain$click_id = NULL
    #valid$click_id = NULL
    
    dtrain = as.data.frame(dtrain)
    valid = as.data.frame(valid)
    test = as.data.frame(test)
    
    dtrain[dtrain == -9999]=NA        
    valid[valid == -9999]=NA 
    test[test == -9999]=NA
    #test = valid[as.integer(runif(10000,1,90000)),]
    
    #######################################################

    
    cat("train size : ", dim(dtrain), "\n")
    cat("valid size : ", dim(valid), "\n")
    
    #######################################################
    
    categorical_features = names(dtrain)[1:10][-c(6,7)]
    
    #######################################################
    invisible(lgb.unloader(wipe = TRUE))
    
    cat("Creating the 'dtrain' for modeling...")
    dvalid = lgb.Dataset(data = as.matrix(dtrain[, colnames(dtrain) != "is_attributed"]), 
                         label = dtrain$is_attributed,categorical_feature = 
                           categorical_features)
    
    #######################################################
    
    cat("Creating the 'dvalid' for modeling...")
    dtrain = lgb.Dataset(data = as.matrix(valid[, colnames(valid) != "is_attributed"]), 
                         label = valid$is_attributed,categorical_feature = 
                           categorical_features)
    
    #######################################################
    
    #rm(valid)
    #invisible(gc())
    
    #######################################################
    
    cat("Modelling")
    
    params = list(objective = "binary", 
                  metric = "auc", 
                  learning_rate= 0.1,
                  num_leaves= 7,
                  max_depth= 4,
                  min_child_samples= 100,
                  max_bin= 100,
                  subsample= 0.7, 
                  subsample_freq= 1,
                  colsample_bytree= 0.7,
                  min_child_weight= 0,
                  min_split_gain= 0,
                  scale_pos_weight= 99.7)
    #######################################################
    
    model = lgb.train(params, dtrain, valids = list(validation = dvalid), nthread = 4,
                      nrounds = 1500, verbose= 1, early_stopping_rounds =100, eval_freq = 25)
    
    #######################################################
    
    rm(dtrain, dvalid)
    invisible(gc())
    
    #######################################################
    maxauc =  max(unlist(model$record_evals[["validation"]][["auc"]][["eval"]]))
    cat("!!!!!!!!!!!!!!!\n Validation AUC @ best iter: ",maxauc, "\n!!!!!!!!!!!!!!!!\n")
    
    #######################################################
    
    #*****************************************************************
    
    
    #*****************************************************************
    # Feature importance
    
    #######################################################
    
    #cat("Feature importance: ")
    #kable(lgb.importance(model, percentage = TRUE))
    fi = lgb.importance(model, percentage = TRUE)
    
    #fi$Feature
    # Predictions
    #train = merge(train, test, all=T)
    nfeat = names(train)
    tokeep = unique(c((1:lkeep),which(nfeat %in% fi$Feature)))
    
    for(i in 1:length(fi$Feature))
    {
      feathash[[fi$Feature[i]]][1] =as.numeric(feathash[[fi$Feature[i]]][1]) + (fi$Gain[i]+fi$Cover[i] + fi$Frequency[i])*ifelse(maxauc>0.965,maxauc,maxauc/3)/3
      feathash[[fi$Feature[i]]][2] =as.numeric(feathash[[fi$Feature[i]]][2]) + 1
      
      if(maxauc>=as.numeric(feathash[[fi$Feature[i]]][3]))
       feathash[[fi$Feature[i]]][3] = as.numeric(maxauc)
    }
    
    train = train[,nfeat[-tokeep]:=NULL]
    prepl = psize*2+length(tnames)-length(tokeep)
    print(paste0("deleting ",prepl," features"))
    
    cat("Begin feature engineering \n")
    i=1
    print(tnames)
    while(i<=prepl/2)
    {
       #print(i)
    a = list()
    intorder = rbinom(n = 1,size = 10,prob = 0.1)+2
    nc=NULL
    for(j in 1:intorder)
    {
      cura = tnames[sample.int(size = 1,n = length(tnames))]
      nc =c(nc,cura)
      a[[cura]]=train[[cura]]
    }

    fnamec = paste0("count_",paste0(sort(unique(nc)),collapse ="_"))
    if(!fnamec %in% names(train))
        invisible(train[, paste0(fnamec):=.N, by=a])
    if(max(train[[fnamec]])==1)
    {
      train[[fnamec]] = NULL
      i=i-1
    }else
    {
      if(!has.key(hash = feathash,key =  fnamec))
        feathash[[fnamec]] = c(as.numeric(0),as.numeric(1),as.numeric(0))
      
      fnameo = paste0("occur_",paste0(sort(unique(nc)),collapse ="_"))
      
      if(!has.key(hash = feathash,key =  fnameo))
        feathash[[fnameo]] = c(as.numeric(0),as.numeric(1),as.numeric(0))
      
      if(!fnameo %in% names(train) && runif(1,0,1)<0.1)
        invisible(train[, paste0(fnameo):=1:.N, by=a])
        
        
      fnameo = paste0("sdhour_",paste0(sort(unique(nc)),collapse ="_"))
      
      if(!has.key(hash = feathash,key =  fnameo))
        feathash[[fnameo]] = c(as.numeric(0),as.numeric(1),as.numeric(0))
      
      if(!fnameo %in% names(train) && runif(1,0,1)<0.9)
        invisible(train[, paste0(fnameo):=sd(hour), by=a])    
        
      fnameo = paste0("meanhour_",paste0(sort(unique(nc)),collapse ="_"))
      
      if(!has.key(hash = feathash,key =  fnameo))
        feathash[[fnameo]] = c(as.numeric(0),as.numeric(1),as.numeric(0))
      
      if(!fnameo %in% names(train) && runif(1,0,1)<0.9)
        invisible(train[, paste0(fnameo):=mean(hour), by=a])     
        
       fnameo = paste0("sdattrib_",paste0(sort(unique(nc)),collapse ="_"))
      
      if(!has.key(hash = feathash,key =  fnameo))
        feathash[[fnameo]] = c(as.numeric(0),as.numeric(1),as.numeric(0))
      
      if(!fnameo %in% names(train) && runif(1,0,1)<0.9)
        invisible(train[, paste0(fnameo):=sd(is_attributed), by=a])    
        
      fnameo = paste0("meanattrib_",paste0(sort(unique(nc)),collapse ="_"))
      
      if(!has.key(hash = feathash,key =  fnameo))
        feathash[[fnameo]] = c(as.numeric(0),as.numeric(1),as.numeric(0))
      
      if(!fnameo %in% names(train) && runif(1,0,1)<0.9)
        invisible(train[, paste0(fnameo):=mean(is_attributed), by=a])     
        
        fnameo = paste0("cumattrib_",paste0(sort(unique(nc)),collapse ="_"))
      
      if(!has.key(hash = feathash,key =  fnameo))
        feathash[[fnameo]] = c(as.numeric(0),as.numeric(1),as.numeric(0))
      
      if(!fnameo %in% names(train) && runif(1,0,1)<0.9)
        invisible(train[, paste0(fnameo):=cumsum(is_attributed), by=a])     
           
        
        if(runif(1,0,1)<0.5)
        { 
            fcura = paste0("lag_appf",paste0(sort(unique(nc)),collapse ="_"))
            
            if(!fcura %in% names(train) && runif(1,0,1)<0.2)
                invisible(train[, paste0(fcura) := c(-9999, app[-length(app)]), by = a])
            if(!has.key(hash = feathash,key = fcura))
            feathash[[fcura]] = c(as.numeric(0),as.numeric(1),as.numeric(0))
        }else{
            
            fcura = paste0("lag_appb",paste0(sort(unique(nc)),collapse ="_"))
             if(!fcura %in% names(train) && runif(1,0,1)<0.2)
            invisible(train[, paste0(fcura) :=c(app[-1],-9999), by = a])
            if(!has.key(hash = feathash,key = fcura))
            feathash[[fcura]] = c(as.numeric(0),as.numeric(1),as.numeric(0))
        }
        
        if(runif(1,0,1)<0.5)
      { 
        fcura = paste0("lag_ipf",paste0(sort(unique(nc)),collapse ="_"))
        if(!fcura %in% names(train) && runif(1,0,1)<0.2)
          invisible(train[, paste0(fcura) := c(-9999, ip[-length(ip)]), by = a])
        if(!has.key(hash = feathash,key = fcura))
          feathash[[fcura]] = c(as.numeric(0),as.numeric(1),as.numeric(0))
      }else{
        fcura = paste0("lag_ipb",paste0(sort(unique(nc)),collapse ="_"))
        if(!fcura %in% names(train) && runif(1,0,1)<0.2)
          invisible(train[, paste0(fcura) :=c(ip[-1],-9999), by = a])
        if(!has.key(hash = feathash,key = fcura))
          feathash[[fcura]] = c(as.numeric(0),as.numeric(1),as.numeric(0))
      }
        
        if(runif(1,0,1)<0.5)
        { 
            fcura = paste0("lag_chf",paste0(sort(unique(nc)),collapse ="_"))
             if(!fcura %in% names(train) && runif(1,0,1)<0.2)
            invisible(train[, paste0(fcura) := c(-9999, channel[-length(channel)]), by = a])
            if(!has.key(hash = feathash,key = fcura))
            feathash[[fcura]] = c(as.numeric(0),as.numeric(1),as.numeric(0))
        }else{
            fcura = paste0("lag_chb",paste0(sort(unique(nc)),collapse ="_"))
            invisible(train[, paste0(fcura) :=c(channel[-1],-9999), by = a])
            if(!has.key(hash = feathash,key = fcura))
            feathash[[fcura]] = c(as.numeric(0),as.numeric(1),as.numeric(0))
        }
        
        if(runif(1,0,1)<0.5)
        { 
            fcura = paste0("lag_osf",paste0(sort(unique(nc)),collapse ="_"))
             if(!fcura %in% names(train) && runif(1,0,1)<0.2)
            invisible(train[, paste0(fcura) := c(-9999,os[-length(os)]), by = a])
            if(!has.key(hash = feathash,key = fcura))
            feathash[[fcura]] = c(as.numeric(0),as.numeric(1),as.numeric(0))
        }else{
            fcura = paste0("lag_osb",paste0(sort(unique(nc)),collapse ="_"))
             if(!fcura %in% names(train) && runif(1,0,1)<0.2)
            invisible(train[, paste0(fcura) :=c(os[-1],-9999), by = a])
        if(!has.key(hash = feathash,key = fcura))
            feathash[[fcura]] = c(as.numeric(0),as.numeric(1),as.numeric(0))
        }
        
        if(runif(1,0,1)<0.5)
        { 
            fcura = paste0("lag_devf",paste0(sort(unique(nc)),collapse ="_"))
             if(!fcura %in% names(train) && runif(1,0,1)<0.2)
            invisible(train[, paste0(fcura) := c(-9999,device[-length(device)]), by = a])
            if(!has.key(hash = feathash,key = fcura))
            feathash[[fcura]] = c(as.numeric(0),as.numeric(1),as.numeric(0))
        }else{
            fcura = paste0("lag_devb",paste0(sort(unique(nc)),collapse ="_"))
             if(!fcura %in% names(train) && runif(1,0,1)<0.2)
            invisible(train[, paste0(fcura) :=c(device[-1],-9999), by = a])
            if(!has.key(hash = feathash,key = fcura))
            feathash[[fcura]] = c(as.numeric(0),as.numeric(1),as.numeric(0))
        }
        
        if(runif(1,0,1)<0.5)
        { 
            fcura = paste0("lag_timef",paste0(sort(unique(nc)),collapse ="_"))
             if(!fcura %in% names(train) && runif(1,0,1)<0.2)
            invisible(train[, paste0(fcura) := c(-9999,time[-length(time)]), by = a])
            if(!has.key(hash = feathash,key = fcura))
            feathash[[fcura]] = c(as.numeric(0),as.numeric(1),as.numeric(0))
        }else{
            fcura = paste0("lag_timeb",paste0(sort(unique(nc)),collapse ="_"))
             if(!fcura %in% names(train) && runif(1,0,1)<0.2)
            invisible(train[, paste0(fcura) :=c(time[-1],-9999), by = a])
        if(!has.key(hash = feathash,key = fcura))
            feathash[[fcura]] = c(as.numeric(0),as.numeric(1),as.numeric(0))
        }
        
         
    }
    i = i+1
    
      
      
    }
    
    
    #######################################################
    #test = as.data.frame(test)
    #test$is_attributed=NULL
    #test$click_time=NULL
    cat("Predictions: \n")
    
    preds = predict(model, data = as.matrix(test[, colnames(test)[-c(6,7)]]), n = model$best_iter)
    
    cat("Setting up the submission file... \n")
    sub = data.table(click_id = test$click_id, is_attributed = NA) 
    invisible(gc())
    cat("Creating the submission data: \n")
    sub$is_attributed = preds
    
    cat("Removing test... \n")
    rm(test)
    invisible(gc())
    
    #######################################################
    sub =  sub[order(click_id)] 
    
    cat("Rounding: \n")
    sub$is_attributed = round(sub$is_attributed, 9)
    
    #######################################################
    
    cat("Writing into a csv file: \n")
    fwrite(sub, paste0(d,"_","results","_",max(unlist(model$record_evals[["validation"]][["auc"]][["eval"]])),".csv"))
    #cat("Converting to data frame: \n")
    #preds = as.data.frame(preds)
    #test_rows = testing_size*0.1
    #skip_rows_test = test_rows*(d-1)
    #18790470
    #if(maxauc>=bestauc)
    #{
        
    #    lid = length(test$click_id)
    #    res[(skip_rows_test+1):(skip_rows_test + lid ),1] = test$click_id
    #    res[(skip_rows_test+1):(skip_rows_test + lid ),2] = preds
    #    res[(skip_rows_test+1):(skip_rows_test + lid ),3] = rep(maxauc,lid)
    #    bestauc = maxauc
       
    #}
    #######################################################
    

    
    #######################################################
    
    #cat("Removing test... \n")
    #rm(test)
    #invisible(gc())
    
    #######################################################
    
    #cat("\nAll done!..")
    #ord = order(values(feathash), decreasing = T)
    #featmap = cbind(keys(feathash)[ord],values(feathash)[ord])
    #print(cbind(keys(feathash)[ord],values(feathash)[ord]))
    
    #this file will output importance statistics upon all the explored features
    #fwrite(as.data.frame(featmap) , paste0(d,"_",g,"featimportance.csv"))
    
  }
  rm(train)
  invisible(gc())
}
ranks = values(feathash)[1,]/values(feathash)[2,]
ord = order(ranks, decreasing = T)
featmap = cbind(keys(feathash)[ord],values(feathash)[3,ord],values(feathash)[1,ord],values(feathash)[2,ord],ranks[ord])
 
featmap = as.data.frame(featmap) 
names(featmap) = c("feature","maxauc","sumimportance","numpresent","mean importance")
#print(cbind(keys(feathash)[ord],values(feathash)[ord]))

#this file will output importance statistics upon all the explored features
fwrite(featmap , paste0("0finalfeatimportance.csv"))

#this file will output importance statistics upon all the explored features
#fwrite(as.data.frame(res) , paste0("allres.csv"))

#cat("Setting up the submission file... \n")
#sub = data.table(click_id = res[,1], is_attributed =res[,2]) 
#invisible(gc())

#sub$is_attributed = round(sub$is_attributed, 9)
#sub =  sub[order(click_id)] 
#######################################################

##cat("Writing into a csv file: \n")
#fwrite(sub, "locallyoptimalsubmission.csv")
