{"cells":[{"metadata":{"_uuid":"a4b4f97514e994192964fe9a18fc2d410871753a","_execution_state":"idle","trusted":true,"_kg_hide-output":true},"cell_type":"code","source":"library(tidyverse)\nlibrary(jsonlite)\nlibrary(stringr)\nlibrary(feather)\nlibrary(magrittr)\nlibrary(lubridate)\nlibrary(keras)\nlibrary(xgboost)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"0421c5f59855a1daba0037901d10ad451df15685"},"cell_type":"code","source":"col_types <- cols(\n  channelGrouping = col_character(),\n  customDimensions = col_character(),\n  date = col_datetime(), # Parses YYYYMMDD\n  device = col_character(),\n  fullVisitorId = col_character(),\n  geoNetwork = col_character(),\n  hits = col_skip(), # MASSIVE amount of data!\n  #sessionId = col_character(), # not present in v2 comp; not used anwyay\n  socialEngagementType = col_skip(), # Skip as always \"Not Socially Engaged\"\n  totals = col_character(),\n  trafficSource = col_character(),\n  visitId = col_integer(), # visitId & visitStartTime look identical in all but 5000 cases\n  visitNumber = col_integer(),\n  visitStartTime = col_integer() # Convert to POSIXlt later,\n  )","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"5f6afc3a247545b00ec119f8f4cfd07a55f6f1c1"},"cell_type":"markdown","source":"# Convert Python array/dictionary string to JSON format"},{"metadata":{"trusted":true,"_uuid":"644525738402088d68f4d47918c7f5c368eef967"},"cell_type":"code","source":"unsnake <- . %>%\n  str_replace_all(c(\"\\\\[\\\\]\" = \"[{}]\", # empty element must contain dictionary\n                    \"^\\\\[|\\\\]$\" = \"\", # remove initial and final brackets\n                    \"(\\\\[|\\\\{|, |: |', )'\" = \"\\\\1\\\"\", # open single- to double-quote (on key or value)\n                    \"'(\\\\]|\\\\}|: |, )\" = '\\\"\\\\1')) # close quote\n\nseparate_json <- . %>%\n    str_replace_all(c(\"\\\"[A-Za-z]+\\\": \\\"not available in demo dataset\\\"(, )?\" = \"\",\n                      \", \\\\}\" = \"}\")) %>% # if last property in list was removed\n    paste(collapse = \",\") %>% paste(\"[\", ., \"]\") %>% # As fromJSON() isn't vectorised\n    fromJSON(., flatten = TRUE)\n\nNMAX = Inf\ndf <- \n  bind_rows(\n    read_csv(\"../input/train_v2.csv\", col_types = col_types, n_max = NMAX) %>% mutate(test = F),\n    read_csv(\"../input/test_v2.csv\",  col_types = col_types, n_max = NMAX) %>% mutate(test = T)\n    ) %>%\n  bind_cols(separate_json(.$device))        %>% select(-device) %>%\n  bind_cols(separate_json(.$geoNetwork))    %>% select(-geoNetwork) %>%\n  bind_cols(separate_json(.$totals))        %>% select(-totals) %>%\n  bind_cols(separate_json(.$trafficSource)) %>% select(-trafficSource) %>%\n  bind_cols(separate_json(unsnake(.$customDimensions))) %>% select(-customDimensions)\n","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"5bda9546d9f708627cb17c5c464e65108e9b0511"},"cell_type":"markdown","source":"# Remove junk"},{"metadata":{"trusted":true,"_uuid":"39f16c1aadfcbde3e6d3b29bbe2fdba6e7d82f76"},"cell_type":"code","source":"df$visits <- NULL # [2] constant column, no information\ndf$adwordsClickInfo.gclId <- NULL #  no obvious information","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"48952506aa7f9970512564ef6a4fd919a1b5a3a4"},"cell_type":"markdown","source":"# Identify types and save as feather (DataFrame)\n"},{"metadata":{"trusted":true,"_uuid":"f788482183fdb147723691e3d0f343f6ae56f606"},"cell_type":"code","source":"df <-\n  df %>%\n  mutate_at(vars(hits:transactions, index), as.integer) %>%\n  mutate(\n    visitStartTime = lubridate::as_datetime(visitStartTime),\n    transactionRevenue = as.numeric(transactionRevenue), # Target\n    totalTransactionRevenue = as.numeric(transactionRevenue)\n    )\n\nformat(object.size(df), units = \"auto\")\n\nwrite_feather(df, \"df.feather\")","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"a980488fb47c0d77383f9c901b66a1d4ca79bab8"},"cell_type":"markdown","source":"# Data loading and adding the isNA column (check trasnsactionRevenue)"},{"metadata":{"trusted":true,"_uuid":"6aa67f8de101440e26c3b0442ab7961fedb8668a"},"cell_type":"code","source":"df <- read_feather('df.feather')%>% select(-totalTransactionRevenue)\ndf <- cbind(df, isNA=1)\ndf$isNA[is.na(df$transactionRevenue)] <- 0","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"5195939a6fbb1e9dee462ccec947a3213300842a"},"cell_type":"markdown","source":"# Predict the fullVisitorId transactionRevenue using Keras\n    * Data culation for XGboost, and adding mean values of pageview "},{"metadata":{"trusted":true,"_uuid":"421673c78c54d15af5ecf946d513b64bed9690ff"},"cell_type":"code","source":"grp_mean <- function(x, grp) ave(x, grp, FUN = function(x) mean(x, na.rm = TRUE))\n\ndata_culation_for_keras <- . %>% \n  mutate(wday = wday(date) %>% factor(),\n         hour = hour(as_datetime(visitStartTime)) %>% factor(),\n         isMobile = ifelse(isMobile, 1L, 0L),\n         isTrueDirect = ifelse(isTrueDirect, 1L, 0L),\n         adwordsClickInfo.isVideoAd = ifelse(!adwordsClickInfo.isVideoAd, 0L, 1L)) %>% \n  select(-date, -fullVisitorId, -visitId, -hits, -visitStartTime, -campaignCode) %>% \n  mutate_if(is.character, factor) %>% \n  mutate(pageviews_mean_vn = grp_mean(pageviews, visitNumber),\n         pageviews_mean_country = grp_mean(pageviews, country),\n         pageviews_mean_city = grp_mean(pageviews, city),\n         pageviews_mean_dom = grp_mean(pageviews, networkDomain),\n         pageviews_mean_ref = grp_mean(pageviews, referralPath)) %>% \n  mutate_if(is.factor, fct_explicit_na) %>% \n  mutate_if(is.numeric, funs(ifelse(is.na(.), 0L, .))) %>% \n  mutate_if(is.factor, fct_lump, prop = 0.05) %>% \n  select(-adwordsClickInfo.isVideoAd, )","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"246993e1f43c2f5355482b1182cb07f36308c51f"},"cell_type":"code","source":"df_keras <- df %>% data_culation_for_keras\n                                 \ndata_culation_for_keras2 <- . %>% \n  select(-transactionRevenue, -isNA, -test) %>%\n  model.matrix(~.-1, .) %>% \n  scale() %>% \n  round(4)\n\ny <- df_keras$transactionRevenue[!(df_keras$test)]\nisNA <- df_keras$isNA[!(df_keras$test)]\nisTest <- df_keras$test\n\ndf_keras %<>% data_culation_for_keras2()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ceafba3ce3be871281da7c7a86067b25103552b6"},"cell_type":"code","source":"tr_keras <- df_keras[!isTest,]\nte_keras <- df_keras[ isTest,]","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"0a96dad409988f61030af63ff50a137dabd394ac"},"cell_type":"markdown","source":"# Modeling "},{"metadata":{"trusted":true,"_uuid":"5790b8607d438760b8ac3c8381e4f0c172c61107"},"cell_type":"code","source":"set.seed(0)\nisNAind <- sample(which(isNA==0), sum(isNA==1))\ntr_keras <- rbind(tr_keras[isNAind,], tr_keras[isNA==1,])\nisNAkeras <- data.frame(isNA=c(isNA[isNAind], isNA[isNA==1]))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"35670e99a77dcdfc5589e4fdd48249e49233286c"},"cell_type":"code","source":"m_nn <- keras_model_sequential() \nm_nn %>% \n  layer_dense(units = 256, activation = \"relu\", input_shape = ncol(tr_keras)) %>% \n  layer_dropout(rate = 0.4) %>% \n  layer_dense(units = 128, activation = \"relu\") %>%\n  layer_dropout(rate = 0.3) %>%\n  layer_dense(units = 2, activation = \"softmax\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"0ec64e50e14eb102dd122e99afd37204c342ba81"},"cell_type":"code","source":"m_nn %>% compile(\n    loss = \"categorical_crossentropy\",\n    optimizer = optimizer_rmsprop(),\n    metrics = c(\"accuracy\")\n)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a6cea26451ff415061780d6e65208955671cc830"},"cell_type":"code","source":"history <- m_nn %>% \n  fit(tr_keras, to_categorical(isNAkeras[,1],2), \n      epochs = 100, \n      batch_size = 100, \n      verbose = 0, \n      validation_split = 0.2,\n      callbacks = callback_early_stopping(patience = 5))\npredicted_isNA <- predict(m_nn, te_keras) %>% round()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f0d52fa7304b1d49fda05e9514814e07e68e9318"},"cell_type":"code","source":"result_keras <- data.frame(fullVisitorId=df[isTest,\"fullVisitorId\"],\n                           isNA=predicted_isNA[,2])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ea7cb6edc35c66292b750189fbd43cf5224c2689"},"cell_type":"code","source":"sub <- \"keras_isNA_v2.csv\"\nresult_keras %>%\n  group_by(fullVisitorId) %>%\n  summarise(isNA = log1p(sum(isNA))) %>% \n  mutate(fullVisitorId=as.character(fullVisitorId)) %>%\n  right_join(\n    read_csv(\"../input/sample_submission_v2.csv\"), \n    by = \"fullVisitorId\") %>% \n  mutate(PredictedLogRevenue = round(isNA, 5)) %>% \n  select(-isNA) %>% \n  write_csv(sub)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ba51cd08bc06a811605bd3192938c92ad4426d21"},"cell_type":"code","source":"te_isNA <- predicted_isNA[,2]==1","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"dd9adeefe093115f0ecc82cd8b5da8ce3ce08a82"},"cell_type":"markdown","source":"# Predict the transactionRevenue using xgboost"},{"metadata":{"trusted":true,"_uuid":"ce7c90984fa2d91cd6bc91c13c6cf72888b10677"},"cell_type":"code","source":"tr_xgb <- df_keras[!isTest,][isNA==1,]\nte_xgb <- df_keras[ isTest,][te_isNA,]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"be0a8161f5281cbcf23851a50b1c6ae935c417b0"},"cell_type":"code","source":"dtest <- xgb.DMatrix(data = data.matrix(te_xgb))\ndtrain <- xgb.DMatrix(data = data.matrix(tr_xgb),\n                      label = log1p(y[isNA==1]))\n\np <- list(objective = \"reg:linear\",\n          booster = \"gbtree\",\n          eval_metric = \"rmse\",\n          nthread = 4,\n          eta = 0.05,\n          max_depth = 7,\n          min_child_weight = 5,\n          gamma = 0,\n          subsample = 0.8,\n          colsample_bytree = 0.7,\n          colsample_bylevel = 0.6)\n\nset.seed(0)\nm_xgb <- xgboost(params=p,\n                 data=as.matrix(tr_xgb),\n                 label=log1p(y[isNA==1]),\n                 nrounds=8000)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"51886393239211533928bcfba24d3bb8128a61de"},"cell_type":"code","source":"pred_xgb <- predict(m_xgb, dtest)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"ad851c33b07a1bed4b04884a6c92a11a46a4cda9"},"cell_type":"markdown","source":"# Final Step - Stacking "},{"metadata":{"trusted":true,"_uuid":"6112882cdf3607bfd5647bb993ea23a1dd9c3878"},"cell_type":"code","source":"te <- subset(df, test==T)\nte_pred_NA <- data.frame(fullVisitorId=as.vector(te$fullVisitorId)[!te_isNA],\n                    y=0)\nte_pred_noNA <- data.frame(fullVisitorId=as.vector(te$fullVisitorId)[te_isNA],\n                    y=expm1(pred_xgb))\nte_pred_submit <- rbind(te_pred_NA, te_pred_noNA)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"7051cb5a01f82b546b5269bbe10157613511d49e"},"cell_type":"code","source":"sub <- \"keras_xgb_v2.csv\"\nte_pred_submit %>%\n  group_by(fullVisitorId) %>% \n  summarise(y = log1p(sum(y))) %>% \n  mutate(fullVisitorId=as.character(fullVisitorId)) %>%\n  right_join(\n    read_csv(\"../input/sample_submission_v2.csv\"), \n    by = \"fullVisitorId\") %>% \n  mutate(PredictedLogRevenue = round(y, 5)) %>% \n  select(-y) %>% \n  write_csv(sub)","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"R","language":"R","name":"ir"},"language_info":{"mimetype":"text/x-r-source","name":"R","pygments_lexer":"r","version":"3.4.2","file_extension":".r","codemirror_mode":"r"}},"nbformat":4,"nbformat_minor":1}