---
title: "quake pipe R"
output: html_document
---

```{r Libraries, message=FALSE, warning=FALSE}

library(data.table)
library(dplyr)
library(xgboost)
options(digits = 10)

```

## Find quake breaks

```{r Find quakes chunk by chunk}

train_vars  <- fread("../input/train.csv", nrows = 0)
chunk_size  <- 10 ^ 8
quake_break <- vector("list")

for (i in c(1:7)){
    
    row_offset <- (i - 1) * chunk_size
    
    train <- fread("../input/train.csv",
                    select = c(2),
                    skip = row_offset,
                    nrows = chunk_size,
                    col.names = names(train_vars)[2])
    
    quake_break[[i]] <-
      which(diff(train$time_to_failure) > 0) + (row_offset)
      
    rm(train)
}

quake_break <- unlist(quake_break)
print(quake_break)

```

## Process train data

```{r Process data to segment summaries, quake by quake}

make_features <- function(.df) {

    summarise(.df,
              x_min  = min(acoustic_data),
              x_max  = max(acoustic_data),
              x_05   = quantile(acoustic_data, 0.05),
              x_50   = quantile(acoustic_data, 0.50),
              x_95   = quantile(acoustic_data, 0.95),
              x_mean = mean(acoustic_data),
              x_sd = sd(acoustic_data),
              x_range = x_max - x_min)
    
}

quakes <- length(quake_break) - 1
quake_segs <- vector("list")

for (i in c(1:quakes)){

  quake_single <-
    fread("../input/train.csv",
          skip = quake_break[i] + 1,
          nrows = quake_break[i + 1] - quake_break[i],
          col.names = names(train_vars))

  quake_single <- 
    quake_single %>%
    mutate(seg = (row_number() %/% 150000) + 1)

  seg_summary <-
    quake_single %>%
    group_by(seg) %>%
    make_features()

  row_stats <-
    quake_single %>%
    group_by(seg) %>%
    summarise(rows = n(),
              y = min(time_to_failure))

  seg_summary <-
    seg_summary %>%
    left_join(row_stats, by = "seg") %>%
    mutate(quake_n = i)

  quake_segs[[i]] <- seg_summary
  
}

train_segs <- bind_rows(quake_segs)

head(train_segs)

```

## Basic model

```{r xgboost model}

larger_train_segs <- filter(train_segs, rows > 100000)
X <- as.matrix(select(larger_train_segs, starts_with("x")))
Y <- pull(larger_train_segs, y)

param <- list(max_depth = 1,
              eta = 0.05,
              verbose = T,
              eval_metric = "mae")

xgb.cv(data = X,
       label = Y,
       nfold = 5,
       nrounds = 1000,
       print_every_n = 100,
       params = param)

m1 <- xgboost(data = X,
              label = Y,
              nrounds = 1000,
              print_every_n = 200,
              params = param)

```

## Apply to test segments

```{r apply to test}

test_name <- list.files(path = "../input/test")
test_data <- lapply(test_name, function(x) fread(paste0("../input/test/", x)))
test_feat <- lapply(test_data, function(x) make_features(x))
test_segs <- bind_rows(test_feat)
test_segs <- mutate(test_segs, seg_id = gsub(".csv", "", test_name))
  
X_test    <- as.matrix(select(test_segs, starts_with("x")))
test_segs <- mutate(test_segs, time_to_failure = predict(m1, X_test))

fwrite(select(test_segs, seg_id, time_to_failure), file = "sub.csv")

summary(test_segs$time_to_failure)

```

