# 0 - funcs -----

suppressPackageStartupMessages(library(dplyr))
library(foreach)

# 0 - 1 get patient data -----
get_patient_data <- function(id, train)
{
  df <- dplyr::select(dplyr::filter(train, Patient == id), c(Weeks, FVC))
  df$log_Weeks <- log(13 + df$Weeks) # the relative timing of FVC measurements (varies widely)
  df$log_FVC <- log(df$FVC) # transformed response variable
  df$Patient <- id
  return(df)
}

# 0 - 2 fit, predict and plot -----
fit_predict <- function(df, lambda = NULL,
                        cv = TRUE, plot_=TRUE,
                        legend_ = FALSE)
{
  min_week <- 13
  n <- nrow(df)
  id <- df$Patient[1]

  test_seq_week <- seq(-12, 133)
  log_test_seq_week <- log(min_week + test_seq_week)


  # Fit a smoothing spline, using Leave-one-out cross-validation for regularization
  fit_obj <- stats::smooth.spline(x = df$log_Weeks,
                                  y = df$log_FVC,
                                  cv = cv,
                                  spar = lambda)

  resids <- residuals(fit_obj)
  mean_resids <- mean(resids)
  conf <- max(exp(sd(resids)), 70) # https://www.kaggle.com/c/osic-pulmonary-fibrosis-progression/overview/evaluation

  preds <- predict(fit_obj, x=log_test_seq_week)

  res <- list(Weeks_pred = test_seq_week, FVC_pred = exp(preds$y), id=id, conf=conf)
  conf_sqrt_n <- conf/sqrt(n)
  ubound <- res$FVC_pred + mean_resids + 1.96*conf_sqrt_n # strong hypothesis
  lbound  <- res$FVC_pred + mean_resids - 1.96*conf_sqrt_n


  if (plot_)
  {
    leg.txt <- c("Measured FVC", "Interpolated/Extrapolated FVC", "95% Confidence interval bound")

    plot(df$Weeks, df$FVC, col="blue", type="l", lwd=3,
         xlim = c(-12, 133),
         ylim = c(min(min(lbound), min(df$FVC)),
                  max(max(ubound), max(df$FVC)) ),
         xlab = "Week", ylab = "FVC",
         main = paste0("Patient: ", id))
    lines(res$Weeks_pred, res$FVC_pred)
    lines(res$Weeks_pred, ubound, lty=2, col="red")
    lines(res$Weeks_pred, lbound, lty=2, col="red")
    abline(v = max(df$Weeks), lty=2)
    if(legend_)
      legend("topright", legend = leg.txt,
           lwd=c(3, 1, 1), lty=c(1, 1, 2),
           col = c("blue", "black", "red"))
  }

  return(invisible(list(res = res,
                        spar = fit_obj$spar,
                        mean = res$FVC_pred,
                        ubound = ubound,
                        lbound = lbound,
                        resids = resids)))
}

# import data
train <- read.csv("../input/osic-pulmonary-fibrosis-progression/train.csv")
sample_submission <- read.csv("../input/osic-pulmonary-fibrosis-progression/sample_submission.csv")

(ids <- unique(train$Patient))
(n_patients <- length(ids))
(n_lam <- 14)
(lam_seq <- c(NULL, seq(0.01, 1, length.out = n_lam)))


# 1 - plot -----

# loop on spar
# loop on patients
i <- 152
(df <- get_patient_data(id=ids[i], train))
# Obtain FVC forecasts, with 95% confidence interval
# warnings when repeated measures in the same week
 par(mfrow=c(4, 2))
 for (lam in lam_seq[1:8])
 {
   suppressWarnings(fit_obj <- fit_predict(df, plot_=TRUE,
                                           lambda = lam))
 }

# 2 - submission file -----

i <- 0 # choice of lambda

sub_lambda <- suppressWarnings(foreach::foreach(j = 1:n_patients,
                             .combine = rbind.data.frame) %do%
{

df <- get_patient_data(id=ids[j], train)

if(i == 0)
{
  suppressWarnings(fit_obj <- fit_predict(df, plot_=FALSE,
                                          lambda = NULL))
} else {
  suppressWarnings(fit_obj <- fit_predict(df, plot_=FALSE,
                                          lambda = lam_seq[i]))
}

df_ <- data.frame(fit_obj$res)
df_$Patient_Week <- paste(df_$id, df_$Weeks_pred, sep = "_")
df_$FVC <- df_$FVC_pred
df_$FVC_pred <- NULL
df_$Confidence <- df_$conf
df_$conf <- NULL
df_$Weeks_pred <- NULL
df_$id <- NULL
as.matrix(df_)
}) # end of loop on patients

cat("\n")
cat("\n")
cat("\n")
cat("Submission, lambda = ", lam_seq[i])
sub <- merge(x=sample_submission, y=sub_lambda, by = "Patient_Week")
sub$FVC <- sub$FVC.y
sub$Confidence <- sub$Confidence.y
sub$FVC.x <- NULL
sub$FVC.y <- NULL
sub$Confidence.y <- NULL
sub$Confidence.x <- NULL

print(head(sub))
print(tail(sub))

write.csv(sub, file = "submission.csv",
           row.names = FALSE, quote = FALSE)