# You can write R code here and then click "Run" to run it on our platform

library(dplyr)
library(caret)
library(neuralnet)
library(e1071)

# as good practice, see if there is header and make sure not to import string as factor this is good for date
dftrain <- read.csv("../input/train.csv", header = TRUE, stringsAsFactors = FALSE)
dftest <- read.csv("../input/test.csv", header=TRUE, stringsAsFactors = FALSE)

# check if there is uniform distribution on label to check if resample is needed
# plot shows no imbalance sample is needed
label.freq <- table(dftrain$label)
barplot(label.freq)

# check if there is any missing obs on train data
apply(dftrain, 2, function(x) sum(is.na(x)))
apply(dftest, 2, function(x) sum(is.na(x)))
maxdftrain <- apply(dftrain, 2, max)
mindftrain <- apply(dftrain, 2, min)
# train shows that there are 42K obs with 785 column (784 if label is excluded)
# this is too much data - two options: do more sample, or reduce the features
# also identify several columns who has 0 values for all 42K obs on training
# output shows that there is no datapoint that is too low, so it is okay to do train#

dftrain$label <- as.factor(dftrain$label) 
dftrain.pixel <- dftrain[,2:ncol(dftrain)]
dftest.pixel <- dftest[,2:ncol(dftest)]

#Use caret to preprocess zeroes vector using nearZeroVar and Principle Component Analysis
#nzv Process using Caret
#nzv <- nearZeroVar(dftrain.pixel, saveMetrics = TRUE)		# this command gives all metric
nzv.Default <- nearZeroVar(dftrain.pixel)
dim(dftrain.pixel)
dftrain.pixel.postnzv <- dftrain.pixel[,-nzv.Default]
preProcValues <- preProcess(dftrain.pixel.postnzv, method=c("pca"))
x_trainTransformed <- predict(preProcValues, dftrain.pixel.postnzv)

# apply it on the Test
nzv.DefaultTest <- nearZeroVar(dftest.pixel)
dim(dftest.pixel)
dftest.pixel.postnzv <- dftrain.pixel[,-nzv.Default]
dim(dftest.pixel.postnzv)
dim(dftrain.pixel.postnzv)
preProcValuesTest <- preProcess(dftest.pixel.postnzv, method=c("pca"))
x_testTransformed <- predict(preProcValues, dftest.pixel.postnzv)
dim(x_testTransformed)
dim(x_trainTransformed)

# Because there are many factors, it is probably better to run with random forest
# after Caret PCA, features reduce to be under 100 - faster and better runtime
# Method 1: Random Forest, without cross validation to save runtime and performance
# single run Random forest without using Cross validation = gives validation value of 0.9475
library(randomForest)
samplerows <- sample(1:nrow(dftrain.pixel), nrow(dftrain)*0.8, replace=FALSE)
dftrainsample <- x_trainTransformed[samplerows,]
dftestsample <- x_trainTransformed[-samplerows,]
y_label_sample <- as.factor(dftrain[samplerows,]$label)
rf_sample <- randomForest(dftrainsample, y_label_sample, ntree=500)
# test on validation dataset
y_label_test <- predict(rf_sample, dftestsample)
y_label_test_act <- as.factor(dftrain[-samplerows,]$label)
tmp_acc <- table(y_label_test_act, y_label_test)
ttl_acc <-0
for (i in 1:10) {ttl_acc <- ttl_acc + tmp_acc[i,i] }
rf_acc <- ttl_acc/nrow(dftestsample)
rf_acc


dftrsvm <- cbind.data.frame(dftrain$label, x_trainTransformed)
names(dftrsvm)[1] <- "label"
dftrsvm$label <- as.factor(dftrsvm$label)
samplerows <- sample(1:nrow(dftrain.pixel), nrow(dftrain)*0.8, replace=FALSE)
dftrsvmsample <- dftrsvm[samplerows,]
dftrsvmval <- dftrsvm[-samplerows,]
digit_svm <- svm(label ~ ., data=dftrsvmsample)
y_predict_valsvm <- predict(digit_svm, dftrsvmval)

tmp_acc <- table(y_predict_valsvm, dftrsvmval$label)
ttl_acc <-0
for (i in 1:10) {ttl_acc <- ttl_acc + tmp_acc[i,i] }
svm_acc <- ttl_acc/nrow(dftrsvmval)
svm_acc
# Svm gives training accuracy of 0.9688 (better than Random forest)

y_test_svm <- predict(digit_svm, x_testTransformed)
y_test_svm <- as.data.frame(y_test_svm)
no_id <- 1:nrow(y_test_svm)
svm_digit_test <- cbind.data.frame(no_id, y_test_svm)
names(svm_digit_test)[1] <- "ImageId"
names(svm_digit_test)[2] <- "label"
write.csv(svm_digit_test, "digitsvm.csv", row.names = FALSE, quote = FALSE)

# Method 3: neural network using neuralnet threshold is 2 to finish in 30 min, otherwise takes too long

library(neuralnet)
nnsamplerows <- sample(1:nrow(dftrain.pixel), nrow(dftrain.pixel)*0.8, replace=FALSE)
nntrainsample <- dftrsvm[nnsamplerows,]
nntestsample <-dftrsvm[-nnsamplerows,]

colname <- names(nntrainsample)
strpred <- paste(colname[!colname %in% "label"], collapse = " + ")

alllevel <- levels(nntrainsample$label)
for(tlabel in unique(alllevel)) {
	nntrainsample[paste("lbl", tlabel, sep = "_")] <- ifelse(nntrainsample$label == tlabel, 1, 0)
}
colname <- names(nntrainsample)
stry <- paste(colname[substr(colname,1,4) %in% "lbl_"], collapse = " + ")
nnformula <- as.formula(paste(stry, " ~ ", strpred))
#digit.nn <- neuralnet(nnformula, data=nntrainsample,linear.output=FALSE, hidden=25, threshold = 1, algorithm = "backprop", learningrate = 0.00001, stepmax= 1e3, lifesign = 'full', lifesign.step = 100)
#digit.nn <- neuralnet(nnformula, data=nntrainsample, hidden = 25, linear.output=FALSE, threshold = 1, stepmax = 1e04, err.fct = "ce", lifesign = 'full', lifesign.step = 100)
digit.nn <- neuralnet(nnformula, data=nntrainsample, hidden = c(35,10), linear.output=FALSE, threshold = 25, stepmax = 1e04, err.fct = "ce", lifesign = 'full', lifesign.step = 100)

pr.nn <- compute(digit.nn, nntestsample[,2:92])
nntestsample$pred.nn <- max.col(pr.nn$net.result[,1:10])
nntestsample$pred.nn <- nntestsample$pred.nn -1
nntestsample$pred.nn <- as.factor(nntestsample$pred.nn)
#table(nntestsample$label, nntestsample$pred.nn)
tmp_acc <- table(nntestsample$pred.nn, nntestsample$label)
ttl_acc <-0
for (i in 1:10) {ttl_acc <- ttl_acc + tmp_acc[i,i] }
nn_acc <- ttl_acc/nrow(dftrsvmval)
nn_acc
# this gives 0.89 accuracy with threshold 5 and 10 min run. Try with more threshold

y_test_nn <- compute(digit.nn, x_testTransformed)
y_test_nn <- as.data.frame(y_test_nn)
no_id <- 1:nrow(y_test_nn)
nn_digit_test <- cbind.data.frame(no_id, y_test_nn)
names(nn_digit_test)[1] <- "ImageId"
names(nn_digit_test)[2] <- "label"
write.csv(nn_digit_test, "digitnn.csv", row.names = FALSE, quote = FALSE)

