{"cells":[{"metadata":{},"cell_type":"markdown","source":"I wanted to preprocess the data using the [Recipes package](https://recipes.tidymodels.org/index.html) from [Tidymodels](https://www.tidymodels.org/) because I find it fairly simple to code using this framework. Besides since it is quite verbose, it makes it easy for someone else to read. So I took [this Kernel](https://www.kaggle.com/lgreig/osic-starter-in-r-glmnet) and changed the preprocessing code section for a Tidymodels recipe.\n\nI hope you find it useful for learning purposes. If that is the case, I appreciate the upvote very much.\n\nI might update the code to use even more of the Tidymodels framework packages."},{"metadata":{"trusted":true},"cell_type":"code","source":"library(tidyverse)\nlibrary(recipes)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train <- read_csv('../input/osic-pulmonary-fibrosis-progression/train.csv')\ntest <- read_csv('../input/osic-pulmonary-fibrosis-progression/test.csv')\nsub <- read.csv('../input/osic-pulmonary-fibrosis-progression/sample_submission.csv')","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Feature engineering"},{"metadata":{"trusted":true},"cell_type":"code","source":"full <- bind_rows(train, test) %>% \n  arrange(Patient, Weeks) %>%\n  group_by(Patient) %>%\n  mutate(Initial_Week = first(Weeks),\n         FVC_init = first(FVC),\n         Percent_init = first(Percent),\n         Week_num = Weeks - Initial_Week\n        ) %>%\n  ungroup()\n\ndim(full)\nhead(full)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Preprocessing using recipes package"},{"metadata":{"trusted":true},"cell_type":"code","source":"prepped_rec <- recipe(FVC ~ ., data = full) %>% \n  update_role(Patient, new_role = \"patient id\") %>% \n  step_other(all_predictors(), -all_numeric()) %>%\n  step_zv(all_predictors()) %>%\n  step_dummy(all_predictors(), -all_numeric(), one_hot = TRUE) %>%\n  prep()\n\n# bake() may be used instead of juice(), but since the recipe was prepped with train data, juice does the trick\nfull <- prepped_rec %>% juice()\n\ndim(full)\nhead(full)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"new_features <- c(\"FVC_init\", \"Percent_init\", \"Week_num\", \"Sex_Female\", \"Sex_Male\",\n                  \"SmokingStatus_Currently.smokes\", \"SmokingStatus_Ex.smoker\", \"SmokingStatus_Never.smoked\")\n\ntrain.new <- train %>%\n    left_join(select(full, \"Patient\", \"Weeks\", all_of(new_features)), by = c(\"Patient\", \"Weeks\")) %>%\n    distinct() %>%\n    select(\"Patient\", \"FVC\", \"Age\", all_of(new_features))\n\ntest <- test %>%\n    left_join(select(full, \"Patient\", \"Weeks\", all_of(new_features)), by = c(\"Patient\", \"Weeks\")) %>%\n    distinct() %>%\n    select(\"Patient\", \"FVC\", \"Age\", all_of(new_features))\n\nfull <- full %>% select(\"Patient\", \"FVC\", \"Age\", all_of(new_features))\n\ndim(train.new)\nhead(train.new)\n\ndim(test)\nhead(test)\n\ndim(full)\nhead(full)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#Use a glmnet model to fit data\nlibrary(caret)\n\nset.seed(2)\ntrain.rows <- sample(nrow(train.new), 0.8*nrow(train.new))\ntrain <- train.new[train.rows, ]\nvalidate <- train.new[-train.rows, ]\n\nmodel <- train(FVC ~ ., data = train[, 2:length(train)], method = \"glmnet\",\n               trControl = trainControl(method = \"cv\", number = 5))\nsummary(model)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"coef(model$finalModel, model$finalModel$lambdaOpt)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"model","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# create a metric function for validating accuracy\nmetric <- function(Y_pred, Y_act, sigma){\n    sigma.clipped <- ifelse(sigma>70, sigma, 70)\n    delta <- ifelse(abs(Y_pred- Y_act)>1000, 1000, abs(Y_pred- Y_act))\n    custom.metric <- -(sqrt(2)*delta /sigma.clipped ) - log(sqrt(2)*sigma.clipped)\n    return(mean(custom.metric))\n}","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#Optimise std for metric\n\n#Assume constant variance since this an assumption for the linear model anyway\n\nmetrics <- c()\nfor (i in 1:400){metrics[i]<- metric(train$FVC, predict(model,train), i)}\nstdev.opt <- which(metrics == max(metrics))[1]\n\n#Plot curve\nplot(seq(1, 400), metrics)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#Predict for validate set\nmetric(validate$FVC, predict(model, validate), stdev.opt)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#Build model again - this time on the full set of data\nmodel <- train(FVC ~ ., data = full[, 2:length(full)], method = \"glmnet\",\n               trControl = trainControl(method = \"cv\", number = 5))\n\ncoef(model$finalModel, model$finalModel$lambdaOpt)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#Generate submission file for prediction\nsub_ <- test[1, ]\n\ntest_test = read.csv('../input/osic-pulmonary-fibrosis-progression/test.csv')\n\nfor (i in seq(-12, 133, by=1)){\n    for (j in 1:nrow(test)){\n        new.row <- test[j,]\n        new.row$Week_num <- i - read.csv('../input/osic-pulmonary-fibrosis-progression/test.csv')$Weeks[j]\n        sub_ <-rbind(sub_, new.row)\n    }\n}\n\nsub_ <- sub_[2:nrow(sub_),]\nhead(sub_)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"model","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#predict and attach\npreds <-predict(model, sub_)\nsub$FVC_pred <- preds\nsub$Confidence_pred <-rep(stdev.opt, times=nrow(sub))\nhead(sub)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#Also replace predictions with known data\n#Set confidence to 70\n\nsub[sub$Patient_Week == 'ID00419637202311204720264_6',]$FVC_pred <- test$FVC[1] \nsub[sub$Patient_Week == 'ID00419637202311204720264_6',]$Confidence_pred <- 70\nsub[sub$Patient_Week == 'ID00421637202311550012437_15',]$FVC_pred <- test$FVC[2] \nsub[sub$Patient_Week == 'ID00421637202311550012437_15',]$Confidence_pred <- 70\nsub[sub$Patient_Week == 'ID00422637202311677017371_6',]$FVC_pred <- test$FVC[3] \nsub[sub$Patient_Week == 'ID00422637202311677017371_6',]$Confidence_pred <- 70\nsub[sub$Patient_Week == 'ID00423637202312137826377_17',]$FVC_pred <- test$FVC[4] \nsub[sub$Patient_Week == 'ID00423637202312137826377_17',]$Confidence_pred <- 70\nsub[sub$Patient_Week == 'ID00426637202313170790466_0',]$FVC_pred <-test$FVC[5] \nsub[sub$Patient_Week == 'ID00426637202313170790466_0',]$Confidence_pred <- 70","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#Make submission\nsub$FVC<-sub$FVC_pred\nsub$Confidence<-sub$Confidence_pred\nsub<-sub[, c(1, 2, 3)]\nhead(sub)\n\nwrite.csv(sub, 'submission.csv', row.names = FALSE, quote = FALSE)","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"name":"ir","display_name":"R","language":"R"},"language_info":{"name":"R","codemirror_mode":"r","pygments_lexer":"r","mimetype":"text/x-r-source","file_extension":".r","version":"3.6.3"}},"nbformat":4,"nbformat_minor":4}