{"cells":[{"metadata":{"_uuid":"051d70d956493feee0c6d64651c6a088724dca2a","_execution_state":"idle","trusted":true},"cell_type":"code","source":"library(keras)\nlibrary(caret)\nlibrary(tensorflow)\nlibrary(tidyverse)\nlibrary(reticulate)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# I. RNN"},{"metadata":{"trusted":true},"cell_type":"code","source":"\n\nriii <- import_from_path('competition','../input/riiid-test-answer-prediction/riiideducation/')\n\n\n\n\n\nriid_data <- read.csv(\"../input/riiid-test-answer-prediction/train.csv\", nrows = 10000) \n\nriid_data <- riid_data %>%\nfilter(answered_correctly != \"-1\") \n\n\n# setting model parameters - easier to do this in one place in code \n\n# max length \nmax_len <- 6 \n\n#batch size \nbatch_size <- 32\n\n# total number of epochs to train the model (number of times model will be exposed\n# to whole training set = total number of training examples/batch size) * number of epochs \n\ntotal_epochs <- 15\n\n# set a random seed for reproducibility \nset.seed(123) \n\n\n# Data cleaning and pre-processing\n\nanswer <- riid_data$answered_correctly\ntable(answer) \n\n# sub-sampling to stretch our data (probably not needed in large data set - or not how I have it below at least)\n\nstart_indexes <- seq(1, length(answer) - (max_len + 1), by = 3)\n\n#create an empty matrix to store our data in \nanswer_matrix <- matrix(nrow = length(start_indexes), ncol = max_len + 1)\n\n# fill in matrix with overlapping slices of dataset \n\nfor (i in 1:length(start_indexes)) {\n  answer_matrix[i, ] <- answer[start_indexes[i]: (start_indexes[i] + max_len)] \n}\n\n\n# make sure matrix is numeric \nanswer_matrix <- answer_matrix*1\nclass(answer_matrix) \n\n# remove na's if any are present \n\nif(anyNA(answer_matrix)) { \n  answer_matrix <- na.omit(answer_matrix) \n  }\n\n# split data into prediction, Y (answer correctness), and the timestamp leading up, X\n\nX <- answer_matrix[, -ncol(answer_matrix)]\nlength(X) \ny <- answer_matrix[, ncol(answer_matrix)]\nlength(y)\n\n\n\n# data splits using createDataPartition() function from caret \n\ntraining_index <- createDataPartition(y, p = 0.9, \n                                      list = FALSE, \n                                      times = 1) \n\n# training data\nX_train <- array(X[training_index,], dim = c(length(training_index), max_len, 1))\ny_train <- y[training_index]\n\n# testing data\nX_test <- array(X[-training_index,], dim = c(length(y) - length(training_index), max_len, 1))\ny_test <- y[-training_index]\n\n\n# initialize model \nmodel <- keras_model_sequential()\n\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"dim(X_train)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Specify the number of units (\"neurons\") in the input layer "},{"metadata":{"trusted":true},"cell_type":"code","source":"model %>% layer_dense(input_shape = dim(X_train)[2:3], units = max_len) ","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Add one hidden layer "},{"metadata":{"trusted":true},"cell_type":"code","source":"model %>% layer_simple_rnn(units = max_len) ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"model %>%\nlayer_dense(units = 1, activation = 'sigmoid') ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"summary(model)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"model %>%\ncompile(loss = 'binary_crossentropy', \n       optimizer = 'RMSprop', \n       metrics = c('accuracy')) ","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Train the model "},{"metadata":{"trusted":true},"cell_type":"code","source":"trained_model <- model %>% fit(\nx = X_train, \ny = y_train, \nbatch_size = batch_size, \nepochs = total_epochs, \nvalidation_spolit = 0.1) ","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Evaluate the model "},{"metadata":{"trusted":true},"cell_type":"code","source":"trained_model","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plot(trained_model)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Predict on test set "},{"metadata":{"trusted":true},"cell_type":"code","source":"classes <- model %>% \npredict_classes(X_test, batch_size = batch_size) \n\nhead(classes)\n\noutput <- classes\n\n\n# confusion matrix \ntable(y_test, classes) \n\n\nnew_output <- cbind(riid_data$row_id, classes)\n\ncolnames(new_output) <- c(\"row_id\", \"answered_correctly\") \n\n#write.csv(new_output, \"submission.csv\", row.names = F)\n\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"model %>%\nevaluate(X_test, y_test, batch_size = batch_size) ","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# II. Naive Bayes "},{"metadata":{"trusted":true},"cell_type":"code","source":"#library(naivebayes) ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#riid_data <- riid_data %>%\n # mutate(answer = ifelse(answered_correctly == \"1\", \"yes\", \"no\")) \n\n#mod_bayes <- naive_bayes(answer~ timestamp + content_id, data = riid_data)\n\n#mod_bayes","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# add new data to predict \n\n#new_data <- read.csv(\"../input/riiid-test-answer-prediction/train.csv\", nrows = 100000)\n\n#new_data <- sample(new_data, size = 10000, replace = TRUE) \n\n#new_data <- new_data %>%\n#filter(answered_correctly != \"-1\") \n\n#new_data <- new_data %>%\n  #mutate(answer = ifelse(answered_correctly == \"1\", \"yes\", \"no\")) \n\n#predictions <- (predict(mod_bayes, new_data, type = \"prob\"))\n\n#head(predictions)\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# data splits \n\n\nset.seed(401) \n\n\nindex_training <- createDataPartition(riid_data$answered_correctly, p = 0.5, list = FALSE) \ntraining_set <- riid_data[index_training, ]\ntesting_set <- riid_data[-index_training, ]\n\n\n\n# logistic regression \n\nmulti_log <- glm(answered_correctly ~ content_id + task_container_id\n                 + prior_question_elapsed_time,\n                 data = training_set, family = binomial)\n\n\ntraining_prediction <- predict(multi_log, \n                               newdata = training_set, type = \"response\") \n\nhist(training_prediction)\n\ntesting_prediction <- predict(multi_log,\n                              newdata = testing_set, type = \"response\") \n\n\nprediction_cutoff <- ifelse(testing_prediction > 0.5, 1, 0) \n\n\nconfusionMatrix(as.factor(prediction_cutoff), as.factor(testing_set$answered_correctly))\n\nhead(testing_set)\n\nlogit_output <- cbind(testing_set$row_id, prediction_cutoff)\n\ncolnames(logit_output) <- c(\"row_id\", \"answered_correctly\")\n\nhead(logit_output)\n\nwrite.csv(logit_output, \"submission.csv\", row.names = F)\n\n\n\n\n","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"name":"ir","display_name":"R","language":"R"},"language_info":{"name":"R","codemirror_mode":"r","pygments_lexer":"r","mimetype":"text/x-r-source","file_extension":".r","version":"3.6.3"}},"nbformat":4,"nbformat_minor":4}