#This is a script implementing a simple genetic algorithm for frequency and novelty based features automatic engineering and evaluation accross the whole dataset.
#Unfortunatelly I don't have time to properly participate in this competition, but I hope the script I made could be benefitial for the participants to improve the scores.
#The script selects datas = N (currently set to just 5 for time related reasons) random samples from the entire dataset uniformly and for each of the datasets performs giter = M (currently set to just 10 for time related reasons) genetic populations of various frequency based features
#The selection criteria for the features to make it to the next population (population size is psize, currenly set to just 100 for memory and time related reasons) is being present in the set of most important features provided by lgb. Statistics for the overall importance 
#of all of the features is saved in hash table feathash() and is based upon the sum of (fi$Gain[i]+fi$Cover[i] + fi$Frequency[i])*maxauc/3 accross all runs for all datasets (giter*datas) it is not normalized with respect to the complexities of the features and probability of drawing them. 
#Here maxauc is the AUC on the current validation data set for the current run. The features with the highest rankings (hopefully are homogeniously important for the whole data) 
#and can be used on the entire dataset in your final analysis. 

#Final feature importance based on the performed set of runs is saved in 1finalfeatimportance.csv!!!

#I don't have time to improve the script anyhow, but the ways to proceed are:

#1) Add time lagged features in the feature engineering part and some other functional transformations of the features (there is a lot of space for creativity).
#2) Change the survival and quality evalutaion metrics (those are very heuristically defined currently).
#3) If 1:datas covers the whole data, the results output (say best out of giter for a given dataset) can be sticked together based on the ids for the submission. Then the submission will be based on datas.
#locally optimal models (across giter runs). 
#4) Increse the testing_size to get more stable results if you have enough RAM, increase datas and giter if you have enough time

#do not hesitate to ask questions and leave comments if you have any.
#many thanks to the guys from the older puclic scripts, from whom I borrowed parts of the code. 

#GOOD LUCK AND HAVE FUN! :) 

library(dplyr)
library(data.table)
library(magrittr)
library(knitr)
library(tidyverse)
library(lubridate)
library(zoo)
library(DescTools)
library(lightgbm)
library(hash)
#######################################################
# Some frequently used control parameters:
####################################################### 
testing = T
#######################################################
testing_size =  500000
#######################################################
val_ratio = 0.90
#######################################################
####  fit parts of data seprately to save memory  #####
partition = 1   
psize = 100
train_path = "../input/talkingdata-adtracking-fraud-detection/train.csv"
test_path  = "../input/talkingdata-adtracking-fraud-detection/test.csv"
fw_train_deltas_path = "../input/td-forward-time-deltas-as-csv/td_forward_time_deltas.csv"
bw_train_deltas_path = "../input/td-backward-time-deltas-as-csv/td_time_deltas.csv"
fw_test_deltas_path = "../input/td-test-forward-time-deltas-as-csv/td_test_forward_time_deltas.csv"
bw_test_deltas_path = "../input/td-test-backward-time-deltas-as-csv/td_test_time_deltas.csv"

#######################################

train_col_names = c("ip", "app", "device", "os", "channel", 
                    "click_time", "attributed_time", "is_attributed")

#######################################################

most_freq_hours_in_test_data = c("4","5","9","10","13","14")
least_freq_hours_in_test_data = c("6","11","15")

#######################################################

total_rows = 184903890
# A function for processing the train/test data

#######################################################

process = function(df) {
cat("Building new features...\n")
df = df %>% mutate(wday = Weekday(click_time), 
                   hour = Hour(click_time),
                   minute = Minute(click_time),
                   second = Second(click_time),
                   week = Week(click_time, method = c("iso")),
                   isWeekend = IsWeekend(click_time),
                   in_test_hh = ifelse(hour %in% most_freq_hours_in_test_data, 1,
                                       ifelse(hour %in% least_freq_hours_in_test_data, 3, 2))) 
return(as.data.table(df))
}
  
#######################################################

datas = 5 #set to 5 for being quickly feasible on Kaggle server
giter = 10 #set to 10 for being quickly feasible on Kaggle server
run=1
feathash = hash()
for(d in as.integer(runif(datas,1,368)))
{
  
  cat("Begin Processing Dataset: ", d, " rows. \n")
  train_rows = testing_size
  skip_rows_train =  testing_size*(d-1)
  test_rows = testing_size*0.1
  
  
  #######################################################
  
  #*****************************************************************

  #######################################################
  
  #*****************************************************************
  # Preparing the training data
  
  #######################################################
  
  cat("Reading the training data...\n")
  train = fread(train_path, skip = skip_rows_train, nrows = train_rows, colClasses = list(numeric=1:5),
                showProgress = FALSE, col.names = train_col_names) %>% select(-c(attributed_time))
  fw_deltas = fread(fw_train_deltas_path, skip = skip_rows_train*0.1, nrows = train_rows, 
                    colClasses = "numeric", showProgress = FALSE, col.names = "forward_time_delta")
  train = cbind( train, fw_deltas )
  rm(fw_deltas)
  invisible(gc())
  
  
  cat("Reading the test data: ", test_rows, " rows. \n")
  test = fread(test_path, nrows=test_rows, colClasses=list(numeric=2:6), showProgress = FALSE)
  fw_deltas = fread(fw_test_deltas_path, nrows = test_rows, colClasses = "numeric", showProgress = FALSE)
  test = cbind( test, fw_deltas[,"forward_time_delta"])
  rm(fw_deltas)
  invisible(gc())
  
  test$is_attributed = -1
  
  train = merge(train, test, all=T)
  
  rm(test)
  #######################################################
  
  
  
  #######################################################
  
  train = process(train)
  
  g = 0
  i = 1
  
  
  tnames = names(train)[-7]
  
  if(run ==1)
  {
    run = 2
    
    for(nm in tnames)
      feathash[[nm]]= as.numeric(0)
  }  
  
  cat("Begin feature engineering \n")
  while(i<=psize)
  {
    #print(i)
    a = list()
    intorder = rbinom(n = 1,size = 10,prob = 0.1)+2
    nc=NULL
    for(j in 1:intorder)
    {
      cura = tnames[sample.int(size = 1,n = length(tnames))]
      nc =c(nc,cura)
      a[[cura]]=train[[cura]]
    }
    fnamec = paste0("count_",paste0(sort(unique(nc)),collapse ="_"))
    
    train[, paste0(fnamec):=.N, by=a]
    if(max(train[[fnamec]])==1)
    {
      train[[fnamec]] = NULL
      i=i-1
    }else
    {
      if(!has.key(hash = feathash,key =  fnamec))
        feathash[[fnamec]] = as.numeric(0) 
      
      fnameo = paste0("occur_",paste0(sort(unique(nc)),collapse ="_"))
      
      if(!has.key(hash = feathash,key =  fnameo))
        feathash[[fnameo]] = as.numeric(0) 
      
      train[, paste0(fnameo):=1:.N, by=a]
    }
    i = i+1
    
  }
  
  
  for(g in 1:giter)
  {
    invisible(gc())
    cat("Begin iteration: ", i, " rows for dataset.",d," \n")
    #######################################################
    
    cat("The training set has", nrow(train), "rows and", ncol(train), "columns.\n")
    #cat("The column names of the train are: \n")
    #cat(colnames(train), "\n")
    print("The size of the train is: ") 
    print(object.size(train), units = "auto")
    
    #######################################################
    
    #*****************************************************************
    # Modelling
    
    #######################################################
    
    print("The table of class unbalance")
    table(train$is_attributed)
    
    test = train[train$is_attributed==-1,]
    tmp = train[train$is_attributed>=0,]
    #######################################################
    #val_ratio = 0.1
    print("Prepare data for modeling")
    library(caret)
    train.index = createDataPartition(tmp$is_attributed, p = val_ratio, list = FALSE)
    
    #class(train$ip)
    tmp$click_time = NULL  
    #######################################################
    tmp = as.data.frame(tmp)
    dtrain = tmp[ train.index,]
    valid  = tmp[-train.index,]
    
    #test = valid[as.integer(runif(10000,1,90000)),]
    
    #######################################################
    
    rm(tmp)
    invisible(gc())
    
    #######################################################
    
    cat("train size : ", dim(dtrain), "\n")
    cat("valid size : ", dim(valid), "\n")
    
    #######################################################
    
    categorical_features = names(dtrain)[1:14][-6]
    
    #######################################################
    invisible(lgb.unloader(wipe = TRUE))
    
    cat("Creating the 'dtrain' for modeling...")
    dtrain = lgb.Dataset(data = as.matrix(dtrain[, colnames(dtrain) != "is_attributed"]), 
                         label = dtrain$is_attributed,categorical_feature = 
                           categorical_features)
    
    #######################################################
    
    cat("Creating the 'dvalid' for modeling...")
    dvalid = lgb.Dataset(data = as.matrix(valid[, colnames(valid) != "is_attributed"]), 
                         label = valid$is_attributed,categorical_feature = 
                           categorical_features)
    
    #######################################################
    
    #rm(valid)
    #invisible(gc())
    
    #######################################################
    
    cat("Modelling")
    
    params = list(objective = "binary", 
                  metric = "auc", 
                  learning_rate= 0.1,
                  num_leaves= 7,
                  max_depth= 4,
                  min_child_samples= 100,
                  max_bin= 100,
                  subsample= 0.7, 
                  subsample_freq= 1,
                  colsample_bytree= 0.7,
                  min_child_weight= 0,
                  min_split_gain= 0,
                  scale_pos_weight= 99.7)
    #######################################################
    
    model = lgb.train(params, dtrain, valids = list(validation = dvalid), nthread = 4,
                      nrounds = 15000, verbose= 1, early_stopping_rounds =100, eval_freq = 25)
    
    #######################################################
    
    rm(dtrain, dvalid)
    invisible(gc())
    
    #######################################################
    maxauc =  max(unlist(model$record_evals[["validation"]][["auc"]][["eval"]]))
    cat("!!!!!!!!!!!!!!!\n Validation AUC @ best iter: ",maxauc, "\n!!!!!!!!!!!!!!!!\n")
    
    #######################################################
    
    #*****************************************************************
    
    
    #*****************************************************************
    # Feature importance
    
    #######################################################
    
    #cat("Feature importance: ")
    #kable(lgb.importance(model, percentage = TRUE))
    fi = lgb.importance(model, percentage = TRUE)
    
    #fi$Feature
    # Predictions
    train = merge(train, test, all=T)
    nfeat = names(train)
    tokeep = unique(c(1:length(tnames),which(nfeat %in% fi$Feature)))
    
    for(i in 1:length(fi$Feature))
    {
      feathash[[fi$Feature[i]]] =as.numeric(feathash[[fi$Feature[i]]]) + (fi$Gain[i]+fi$Cover[i] + fi$Frequency[i])*maxauc/3
    }
    
    train = train[,nfeat[-tokeep]:=NULL]
    prepl = psize*2+length(tnames)-length(tokeep)
    print(paste0("deleting ",prepl," features"))
    
    cat("Begin feature engineering \n")
    i=1
    while(i<=prepl/2)
    {
      #print(i)
      a = list()
      intorder = rbinom(n = 1,size = 10,prob = 0.1)+2
      nc=NULL
      for(j in 1:intorder)
      {
        cura = tnames[sample.int(size = 1,n = length(tnames))]
        nc =c(nc,cura)
        a[[cura]]=train[[cura]]
      }
      fnamec = paste0("count_",paste0(sort(unique(nc)),collapse ="_"))
      
      train[, paste0(fnamec):=.N, by=a]
      if(max(train[[fnamec]])==1)
      {
        train[[fnamec]] = NULL
        i=i-1
      }else
      {
        if(!has.key(hash = feathash,key =  fnamec))
          feathash[[fnamec]] = as.numeric(0) 
        
        fnameo = paste0("occur_",paste0(sort(unique(nc)),collapse ="_"))
        
        if(!has.key(hash = feathash,key =  fnameo))
          feathash[[fnameo]] = as.numeric(0) 
        
        train[, paste0(fnameo):=1:.N, by=a]
      }
      i = i+1
      
      
    }
    
    
    #######################################################
    test = as.data.frame(test)
    #test$is_attributed=NULL
    #test$click_time=NULL
    cat("Predictions: \n")
    
    preds = predict(model, data = as.matrix(test[, colnames(test)[-c(6,7)]], n = model$best_iter))
    
    #######################################################
    
    cat("Converting to data frame: \n")
    preds = as.data.frame(preds)
    
    #######################################################
    cat("Setting up the submission file... \n")
    sub = data.table(click_id = test$click_id+skip_rows_train, is_attributed = NA) 
    invisible(gc())
    cat("Creating the submission data: \n")
    sub$is_attributed = preds
    
    #######################################################
    
    cat("Removing test... \n")
    rm(test)
    invisible(gc())
    
    #######################################################
    sub =  sub[order(click_id)] 
    
    
    cat("Rounding: \n")
    sub$is_attributed = round(sub$is_attributed, 9)
    
    #######################################################
    
    cat("Writing into a csv file: \n")
    fwrite(sub, paste0(d,"_","results","_",max(unlist(model$record_evals[["validation"]][["auc"]][["eval"]])),".csv"))
    
    #######################################################
    
    cat("\nAll done!..")
    ord = order(values(feathash), decreasing = T)
    featmap = cbind(keys(feathash)[ord],values(feathash)[ord])
    #print(cbind(keys(feathash)[ord],values(feathash)[ord]))
    
    #this file will output importance statistics upon all the explored features
    fwrite(as.data.frame(featmap) , paste0(d,"_",i,"featimportance.csv"))
    
  }
  rm(train)
  invisible(gc())
}

ord = order(values(feathash), decreasing = T)
featmap = cbind(keys(feathash)[ord],values(feathash)[ord])
#print(cbind(keys(feathash)[ord],values(feathash)[ord]))

#this file will output importance statistics upon all the explored features
fwrite(as.data.frame(featmap) , paste0("1finalfeatimportance.csv"))
