{"cells":[{"metadata":{},"cell_type":"markdown","source":"This is a starter R markdown kernel for baseline prediction using a glmnet model for the OSIC competition using only the tabular data supplied.","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"#Get all data and examine","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train <- read.csv('../input/osic-pulmonary-fibrosis-progression/train.csv')\nhead(train)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test <- read.csv('../input/osic-pulmonary-fibrosis-progression/test.csv')\nhead(test)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"sub <- read.csv('../input/osic-pulmonary-fibrosis-progression/sample_submission.csv')\nhead(sub)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#Combine train and test\nfull<-rbind(train, test)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#Generate some features\n\nIS =list()\n\nWeeks_Elapsed<-c()\nFVC<-c()\nPercent<-c()\nInitial_Week<-c()\n\n\nfor (i in 1:nrow(full))\n{\nentry=as.character(full[i, 1])\nif (is.null(IS[[entry]]))\n{\nIS[[entry]]<-c(full[i, 2],full[i, 3], full[i, 4] )\nInitial_Week[i]<- IS[[entry]][1]\nFVC[i]<-  IS[[entry]][2]\nPercent[i]<- IS[[entry]][3]\nWeeks_Elapsed[i]<-full[i, 2] - Initial_Week[i]\n} else {\nInitial_Week[i]<- IS[[entry]][1]\nFVC[i]<-  IS[[entry]][2]\nPercent[i]<- IS[[entry]][3]\nWeeks_Elapsed[i]<-full[i, 2] - Initial_Week[i]\n}\n}\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#Attach features\n\nfull$FVC_init<-FVC\nfull$Percent_init<-Percent\nfull$Week_num<-Weeks_Elapsed\ntail(full)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#Convert the categorical variables into dummy varaibles\nlibrary(onehot)\nencoder<-onehot(full[, c(6, 7)])\noutput<-predict(encoder,full[, c(6, 7)])\nhead(output)\ncolnames(output)<-c('Fe', 'Male', 'CS', 'EX', 'NS')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#Create a new dataframe with all the information\nfull<-cbind(full, output)\nfull$Sex<-NULL\nfull$SmokingStatus<-NULL\nfull$Percent<-NULL\nfull$Weeks<-NULL\nhead(full)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#Use a glmnet model to fit data\n\nlibrary(caret)\nlibrary(dplyr)\n\ntrain.new<-full[1:1549,]\ntest<-full[1550:nrow(full),]\n#train.new$Patient<-NULL\nset.seed(2)\ntrain.rows<-sample(nrow(train.new), 0.8*nrow(train.new))\ntrain<-train.new[train.rows, ]\nvalidate<-train.new[-train.rows, ]\n\nmodel<-train(FVC~ ., data=train[, 2:length(train)], method=\"glmnet\", trControl=trainControl(method=\"cv\", number=5))\nsummary(model)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"coef(model$finalModel, model$finalModel$lambdaOpt)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"model","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# create a metric function for validating accuracy\n\nmetric <- function(Y_pred, Y_act, sigma)\n{\nsigma.clipped<-ifelse(sigma>70, sigma, 70)\ndelta<-ifelse(abs(Y_pred- Y_act)>1000, 1000, abs(Y_pred- Y_act))\ncustom.metric<- -(sqrt(2)*delta /sigma.clipped ) - log(sqrt(2)*sigma.clipped)\nreturn(mean(custom.metric))\n}","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#Optimise std for metric\n\n#Assume constant variance since this an assumption for the linear model anyway\n\nmetrics<-c()\n\nfor (i in 1:400)\n{\nmetrics[i]<- metric(train$FVC, predict(model,train), i)\n}\n\nstdev.opt<-which(metrics==max(metrics))[1]\n\n#Plot curve\nplot(seq(1, 400), metrics)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#Predict for validate set\n\nmetric(validate$FVC, predict(model,validate), stdev.opt)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#Build model again - this time on the full set of data\n\nmodel<-train(FVC~ ., data=full[, 2:length(full)], method=\"glmnet\", trControl=trainControl(method=\"cv\", number=5))\ncoef(model$finalModel, model$finalModel$lambdaOpt)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#Generate submission file for prediction\n\nsub_ <- test[1, ]\n\n\nfor (i in seq(-12, 133, by=1))\n{\n\nfor (j in 1:nrow(test))\n{\nnew.row <- test[j,]\nnew.row$Week_num <- i - read.csv('../input/osic-pulmonary-fibrosis-progression/test.csv')$Weeks[j]\nsub_ <-rbind(sub_, new.row)\n}\n}\n\nsub_ <- sub_[2:nrow(sub_),]\nhead(sub_)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#predict and attach\npreds<-predict(model, sub_)\nsub$FVC_pred<- preds\nsub$Confidence_pred<-rep(stdev.opt, times=nrow(sub))\nhead(sub)\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#Also replace predictions with known data\n#Set confidence to 70\n\nsub[sub$Patient_Week == 'ID00419637202311204720264_6',]$FVC_pred<- test$FVC[1] \nsub[sub$Patient_Week == 'ID00419637202311204720264_6',]$Confidence_pred<- 70\nsub[sub$Patient_Week == 'ID00421637202311550012437_15',]$FVC_pred<- test$FVC[2] \nsub[sub$Patient_Week == 'ID00421637202311550012437_15',]$Confidence_pred<- 70\nsub[sub$Patient_Week == 'ID00422637202311677017371_6',]$FVC_pred<- test$FVC[3] \nsub[sub$Patient_Week == 'ID00422637202311677017371_6',]$Confidence_pred<- 70\nsub[sub$Patient_Week == 'ID00423637202312137826377_17',]$FVC_pred <- test$FVC[4] \nsub[sub$Patient_Week == 'ID00423637202312137826377_17',]$Confidence_pred<- 70\nsub[sub$Patient_Week == 'ID00426637202313170790466_0',]$FVC_pred <-test$FVC[5] \nsub[sub$Patient_Week == 'ID00426637202313170790466_0',]$Confidence_pred<- 70","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#Make submission\n\nsub$FVC<-sub$FVC_pred\nsub$Confidence<-sub$Confidence_pred\nsub<-sub[, c(1, 2, 3)]\nhead(sub)\n\nwrite.csv(sub, 'submission.csv', row.names=F)","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"name":"ir","display_name":"R","language":"R"},"language_info":{"name":"R","codemirror_mode":"r","pygments_lexer":"r","mimetype":"text/x-r-source","file_extension":".r","version":"3.6.3"}},"nbformat":4,"nbformat_minor":4}