{"cells":[{"metadata":{"_uuid":"3c1a23d7dc213e5ffef5dc1b8c7087c433ecc654","_cell_guid":"877c21f5-e1be-4983-9a33-54ac37c835b7"},"cell_type":"markdown","source":"## Introduction\n\n> TalkingData, China’s largest independent big data service platform, covers over 70% of active mobile devices nationwide. \n>  They handle 3 billion clicks per day, of which 90% are potentially fraudulent. Their current approach to prevent click fraud for app \n> developers is to measure the journey of a user’s click across their portfolio, and flag IP addresses who produce lots of clicks, but never\n> end up installing apps. With this information, they've built an IP blacklist and device blacklist.\n\n[Public Score 0.9383](https://www.kaggle.com/blastchar/talkingdata-h2o)\n\n[Public Score 0.9511](https://www.kaggle.com/blastchar/talkingdata-h2o-improve)\n\n## Libraries"},{"metadata":{"_uuid":"b3714a6d1e34076208332082ddc384e27241b730","trusted":false,"_cell_guid":"4d06ae20-1b43-4dc1-8690-97b69b49036d"},"cell_type":"code","source":"# Data manipulation\nsuppressMessages(library(data.table))\nsuppressMessages(library(dplyr))\n\n# graphics capabilities\nsuppressMessages(library(gridExtra))\nsuppressMessages(library(grid))\nsuppressMessages(library(ggplot2))\n\n# h2o modeling kit\nsuppressMessages(library(h2o))\n\n# AuC Evaluation\nsuppressMessages(library(pROC))\n\n# work with dates\nsuppressMessages(library(lubridate))\n\n# work with dates\nsuppressMessages(library(FSelector))","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"3eb8bc5949033e9656e57db8973c95df3c5ad600","_cell_guid":"57458b2f-418c-447b-b0ff-eb26d72e2b63"},"cell_type":"markdown","source":"## Loading data\nFirst Load data and then extract a nice sample for training the model"},{"metadata":{"_uuid":"12560e6473c0c1fba3299a9c6e1eed4e720f6bf9","trusted":false,"_cell_guid":"adbc8451-7a6b-4f11-bedb-203884c61d9c"},"cell_type":"code","source":"# clear all\nrm(list=ls())","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"437c1ab60d2149f46d1bcb7f99c501342b4b0728","trusted":false,"_cell_guid":"c7151001-45e7-4a70-a956-77b9a6b05016"},"cell_type":"code","source":"# load train.csv\ntrain <- fread(\"../input/train.csv\", showProgress=F, colClasses = list(numeric=1:7))","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"50a2c512138724b97d0caf1d83efd0b5c63cc8bc","trusted":false,"_cell_guid":"8c61b044-51f7-4e69-bedb-8fb5bdf3ebf2"},"cell_type":"code","source":"# extract a nice sample of data\nset.seed(7)\nDstTrain <- train[sample(.N, 2500000), -7]","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"aaab3e44a8d9a033573fbe630608e3d1fb90c900","trusted":false,"_cell_guid":"526e5665-3481-419c-ae38-0fd34f1ce909"},"cell_type":"code","source":"rm(\"train\")\ninvisible(gc())","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"afcd5cbaac41af825805421f2acf48ed75ae8112","_cell_guid":"5f3d9f34-981a-4d3f-a710-b13656e7a13f"},"cell_type":"markdown","source":"## Creating Features\n\nCreatin some of the features for the model"},{"metadata":{"_uuid":"0e91b28611e681abc94346ddfe7c1c2d5014b5d6","trusted":false,"_cell_guid":"3ddc06d6-57c9-4159-8fe9-e75a7edbcd94"},"cell_type":"code","source":"# on my sample this give diferent but I guess the other \n# kernels have analysed the full data so i will use this ones\nFreqHour <- c(\"4\",\"5\",\"9\",\"10\",\"13\",\"14\")\nNotFreqHour <- c(\"6\",\"11\",\"15\")\n\n# create all needed features\nCreateFeatures <- function(dstProcess) {\n    \n    # timeline related features\n    # there's only 4 days in the sample data but maybe is also a feature\n    # many clicks in the same day and specialy in the same hour looks suspicious\n    dstProcess <- mutate(dstProcess,\n        Day = day(ymd_hms(dstProcess$click_time)), \n        Hour = hour(ymd_hms(dstProcess$click_time)),\n        PeriodType = ifelse(hour(ymd_hms(dstProcess$click_time)) %in% FreqHour, 1,\n                    ifelse(hour(ymd_hms(dstProcess$click_time)) %in% NotFreqHour, 3, 2))\n    ) \n    \n    # create some counts \n    # many clicks maybe indicates that some are fraudulent \n    setDT(dstProcess)[ ,IP_Day_PeriodType := .N, by = .(ip, Day, PeriodType)]\n    setDT(dstProcess)[ ,IP_Day_Hour := .N, by = .(ip, Day, Hour)]\n    setDT(dstProcess)[, IP_Day_Hour_os := .N, by = .(ip, Day, Hour, os)]\n    setDT(dstProcess)[, IP_Day_Hour_app := .N, by = .(ip, Day, Hour, app)]\n    setDT(dstProcess)[, IP_Day_Hour_app_os := .N, by = .(ip, Day, Hour, app, os)]\n    setDT(dstProcess)[ ,App_Day_Hour := .N, by = .(app, Day, Hour)]\n\n    # drop ip click_time variables\n    dropVars <- -which(names(dstProcess) %in% c('ip', 'click_time','PeriodType'))\n    dstProcess <- as.data.frame(dstProcess)[, dropVars]\n    \n    # return the processed dst\n    return(dstProcess)\n  }","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"2a33b3f32e8aeaf02490c5817eb192ec4561270d","trusted":false,"_cell_guid":"fad3d517-afb4-4a2c-bd4c-69c94b1c6a34"},"cell_type":"code","source":"# create the features for the model\nDstTrain <- CreateFeatures(DstTrain)\nhead(DstTrain)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"2e7de6b284cbec0806972b943b4cf29c9bb1c844","trusted":false,"_cell_guid":"aeb25d7c-7af4-4b3e-81c0-cff962d07395"},"cell_type":"code","source":"# Calculate the chi square statistics \nweights <- chi.squared(is_attributed~., DstTrain[1:10000, ])\n# Select top five variables\nsubset <- cutoff.k(weights, 15)\n# Print the final formula that can be used in classification\nmyFormula <- as.simple.formula(subset, \"is_attributed\")\n# show me the formula to use \nprint(myFormula)\n# reorder\nDstTrain <- DstTrain[c(\"is_attributed\",subset)]","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"933398ad7ed29f4bc18c26a1d8807936e10794e0","_cell_guid":"e0209560-c80a-4d46-9a39-600450ee68de"},"cell_type":"markdown","source":"## Modeling\n\nStart up a 1-node H2O server on your local machine, and allow it to use all CPU cores and up to 1GB of memory:"},{"metadata":{"_uuid":"beef662a58ebbc8df932067cef5667defb200460","trusted":false,"_cell_guid":"1e97e7bc-a348-4ce8-b080-aef8efb82129"},"cell_type":"code","source":"h2o.init(nthreads=-1, max_mem_size=\"8G\")\nh2o.removeAll()    # clean slate - just in case the cluster was already running\nh2o.no_progress()  # Don't show progress bars in RMarkdown output","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"0d617d6248fbbad57d852f58276a69874cbd980e","trusted":false,"_cell_guid":"fe563fee-94ba-4ccf-84a2-5325f6e8e1fa"},"cell_type":"code","source":"# as factor\nDstTrain$is_attributed <- as.factor(DstTrain$is_attributed)\nlevels(DstTrain$is_attributed) = make.names(unique(DstTrain$is_attributed))","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"267978761e851f3a83094db625ec8f272d57a7f8","trusted":false,"_cell_guid":"fdd39979-14e0-43bd-8e83-d9d25ec6df3a"},"cell_type":"code","source":"# convert to h2o frame \nh2o_DstTrain = as.h2o(DstTrain[c(\"is_attributed\",subset)])","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"2ed8dc02e0758966a0517c7745e243b8a2c4e71d","trusted":false,"_cell_guid":"2ca15cc9-660c-43dc-b110-4f74b08c6cf5"},"cell_type":"code","source":"# Identify predictors and response\nresponse <- \"is_attributed\"\npredictors <- setdiff(names(h2o_DstTrain), response)\n\n# Number of CV folds (to generate level-one data for stacking)\ncvfolds <- 5","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"e42a543f997114ce21481202601ed845980a4df6","_cell_guid":"c43da56c-cab0-4b25-bc80-57fbe9b15f70"},"cell_type":"markdown","source":"**XGBoost base models**"},{"metadata":{"_uuid":"847aeb78a19312e5ad8b4f7b4627242845f174cf","trusted":false,"_cell_guid":"abf38f17-4d3b-40f8-91c5-d18b30fe2548"},"cell_type":"code","source":"# Train & Cross-validate another (deeper) XGB-GBM\nmy_xgb2 <- h2o.xgboost(x = predictors,\n                       y = response,\n                       training_frame = h2o_DstTrain,\n                       distribution = \"bernoulli\",\n                       ntrees = 50,\n                       max_depth = 8,\n                       min_rows = 1,\n                       learn_rate = 0.1,\n                       sample_rate = 0.7,\n                       col_sample_rate = 0.9,\n                       nfolds = cvfolds,\n                       fold_assignment = \"Modulo\",\n                       keep_cross_validation_predictions = TRUE,\n                       seed = 1)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"634359a7ad49b2128db65d83d31910b81fdd8d9b","_cell_guid":"0600352f-22cc-4156-ae0d-073f034884f5"},"cell_type":"markdown","source":"## Final submition"},{"metadata":{"_uuid":"2905457ce712d474838199da5aa7d5ca405a02b3","trusted":false,"_cell_guid":"42ab006e-b627-478e-878a-858b490bac0a"},"cell_type":"code","source":"# load test.csv\nDstTest  <- read.csv('../input/test.csv', stringsAsFactors = FALSE, na.strings = c(\"NA\", \"\"))\n# keep the click id for later submition\nDstTestIDs <- DstTest$click_id","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"e53040d99b0b433b634726f2f3653e69b41b8478","trusted":false,"_cell_guid":"ddc71a6f-45db-4cb6-9119-83b50334826f"},"cell_type":"code","source":"# create the features for the model\nDstTest <- CreateFeatures(DstTest)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d9e6f1b243194ef2c86ad8cf71b2cad0e77caac4","trusted":false,"_cell_guid":"18652b1e-8480-4998-82f3-b361125663d0"},"cell_type":"code","source":"# reorder\nDstTest <- DstTest[subset]","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"ca1cbd129ea8d9be82bfec4bab9731f617573a28","trusted":false,"_cell_guid":"27fe62ec-aeca-4fea-99b2-ead7a5009476"},"cell_type":"code","source":"# convert to h2o frame \nh2o_FinalTest = as.h2o(DstTest)\n\n# predict with the model\npredictFinal <- h2o.predict(my_xgb2, h2o_FinalTest)\n\n# convert H2O format into data frame and save as csv\npredictFinal.df <- as.data.frame(predictFinal)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"f5cf433386990c257572a058f7033edaa524e19f","trusted":false,"_cell_guid":"027148b4-4523-4cc2-9ae1-d93f7bc17df9"},"cell_type":"code","source":"# create a csv file for submittion\nResult <- data.frame(click_id = DstTestIDs, is_attributed = predictFinal.df$X1)\nhead(Result,n=5L)\n# write the submition file\nwrite.csv(Result,file = \"Result.csv\",row.names = FALSE)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"89bd65fe5f51d354737d8640985319d04043aa2c","trusted":false,"_cell_guid":"fa8e9d15-685e-4585-947f-e00e9e8b1861"},"cell_type":"code","source":"# shut down virtual H2O cluster\n h2o.shutdown(prompt = FALSE)","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"R","language":"R","name":"ir"},"language_info":{"mimetype":"text/x-r-source","name":"R","pygments_lexer":"r","version":"3.4.2","file_extension":".r","codemirror_mode":"r"}},"nbformat":4,"nbformat_minor":1}