{"cells":[{"metadata":{"_uuid":"0f6d38adde5bbb2cc3f59313daba1bd54299dcf2","_cell_guid":"65862bf1-5a43-4e0d-b37d-127656f675af"},"cell_type":"markdown","source":""},{"metadata":{"_uuid":"2cc64756aff32d39da3b8e300622eb47444bff5b","_cell_guid":"d135b053-c88d-45d8-b00f-746dee505e6f"},"cell_type":"markdown","source":"# Loading Libraries"},{"metadata":{"_execution_state":"idle","_uuid":"72392901e2772b0b1b4e80e12d6c4c7b29c9f415","trusted":true,"_cell_guid":"0fa1fb16-94d0-4f8c-a1d4-538872d16153"},"cell_type":"code","source":"# This R environment comes with all of CRAN preinstalled, as well as many other helpful packages\n# The environment is defined by the kaggle/rstats docker image: https://github.com/kaggle/docker-rstats\n# For example, here's several helpful packages to load in \n\nlibrary(ggplot2) # Data visualization\nlibrary(readr)\nlibrary(data.table)# CSV file I/O, e.g. the read_csv function\nlibrary(e1071)\nlibrary(randomForest)\nlibrary(RWeka)\nlibrary(C50)\nlibrary(class)\nlibrary(neuralnet)\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nsystem(\"ls ../input\")\n\n# Any results you write to the current directory are saved as output.","execution_count":3,"outputs":[]},{"metadata":{"_uuid":"ee7a13d8bbb8e169c190cfdadb7b71c20fa82996","_cell_guid":"62878538-01dd-4b45-a332-9b5d40acedcd"},"cell_type":"markdown","source":"# Loading Data"},{"metadata":{"_uuid":"4cc912ae66acaaefa551ee1d75a3b89901168c04","trusted":true,"_cell_guid":"0adfdce8-a249-41c7-8bd3-3cd58d1d8d97"},"cell_type":"code","source":"# Lets use train data and we will later split it into training and testing\n# Since the data is quite large, this approach can be implemented on larger data with server and cloud\ntrain <- fread(\"../input/train.csv\", showProgress=F)\n\n\nset.seed(0)\ntrain <- train[sample(.N, 3e6), ]","execution_count":47,"outputs":[]},{"metadata":{"_uuid":"268894788729dbc56c1570afcd17a43f65f6634f","trusted":true,"_cell_guid":"c2f942ed-4f13-4fa1-a912-7ab9ac1446e5","scrolled":true},"cell_type":"code","source":"#Checking Train Data\ndim(train)\nhead(train, 3)","execution_count":48,"outputs":[]},{"metadata":{"_uuid":"b385e8cd8d26811f934d535938f2d7cd4310173a","trusted":true,"_cell_guid":"f883aaa7-4926-4c94-ad0f-a16464d74d37"},"cell_type":"code","source":"#Removing Attributed Time\ntrain <- train[,-7]\nhead(train, 3)\n","execution_count":49,"outputs":[]},{"metadata":{"_uuid":"222994c193eb6c7c82b0ec308c2caec3be133a35","trusted":true,"_cell_guid":"68710d6b-0438-4720-99f7-17dd5dcac26c"},"cell_type":"code","source":"summary(train)","execution_count":50,"outputs":[]},{"metadata":{"_uuid":"cc69fb9f4962a4e1269feca9cf18c13c4df7f4a4","trusted":true,"_cell_guid":"232a33cd-47f1-492a-a7bc-b0710264bf9b"},"cell_type":"code","source":"str(train)","execution_count":51,"outputs":[]},{"metadata":{"_uuid":"5e9e3d49dce8c72276d002334ace3359dd801c90","trusted":true,"_cell_guid":"7df642ad-07eb-4bf9-9cdc-f622c4370466"},"cell_type":"code","source":"#Checking for Null Data\nsapply(train, function(y) sum(is.na(y)))\n# The data looks pretty clean","execution_count":52,"outputs":[]},{"metadata":{"_uuid":"8be14a17cd5921e74469fb1b8d6abaf8aa7e1e64","trusted":true,"_cell_guid":"12f1ac27-2a04-4ad2-b795-235f2f162a40"},"cell_type":"code","source":"# Let's Convert the is_attributed column into factor variable\ntrain$is_attributed <- as.factor(train$is_attributed)\nis.factor(train$is_attributed)\n","execution_count":53,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d18f52d62fb1c8911e0833b46fc4ae3fc0a8f6df"},"cell_type":"code","source":"levels(train$is_attributed)","execution_count":54,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"2bf585cefc20b5cc4528c7d986f592b7c7c08382"},"cell_type":"code","source":"table(train$is_attributed)","execution_count":55,"outputs":[]},{"metadata":{"_uuid":"9ee3ba4c92a8f77b4f17d428135035ca02221930","trusted":true,"_cell_guid":"e1a9a0b7-0d92-4ebd-9b0e-03189a144d64"},"cell_type":"code","source":"head(train,3)","execution_count":56,"outputs":[]},{"metadata":{"_uuid":"9e278c51e230e4b876280d6a83bc4a0c17f06a1f","trusted":false,"_cell_guid":"0d3d9a11-7c4c-4207-a881-11f531083856"},"cell_type":"code","source":"# Train Data is Ready for Feautre Engineering","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"3e8c958d638fd7887999abe3a915c8a6ffd1361d","_cell_guid":"a32a78da-edfe-4865-b00b-28772c7f2260"},"cell_type":"markdown","source":"# Feature Engineering"},{"metadata":{"_uuid":"22146d3d75e655863f89b1cd1fc76e8d4d31446f","trusted":true,"_cell_guid":"d288166c-a540-457a-80ce-06a4c748eda0"},"cell_type":"code","source":"##################################################################################################\n## TIME FEATURE.\n## Based on click time the data is bucketed into groups of 1 min, 5 min, 1 hr, 3 hr and 6 hrs\n## Then the mean, max, variance and skewnwss of these bucket's frequencies are prepared\n## Finally these features are added to the existing dataset and returned to caller as \n## added features\n\n#Round down the minutes to a mutliple of mnslot\nbucketInMins <- function (dtList, mnslot) { \n  str <- strptime(dtList,\"%Y-%m-%d %H:%M:%S\")\n  min_st <- as.numeric(format(str, \"%M\"))\n  roundedmins <- floor(min_st/mnslot) * mnslot\n  base <- strptime(dtList, \"%Y-%m-%d %H\")\n  bucket <- base + (roundedmins * 60)\n  return (data.frame(table(bucket)))\n}\n\n#Round down the hours to a mutliple of hrslot\nbucketInHrs <- function (dtList, hrslot) { \n  str <- strptime(dtList,\"%Y-%m-%d %H:%M:%S\")\n  hr_st <- as.numeric(format(str, \"%H\"))\n  roundedhrs <- floor(hr_st/hrslot) * hrslot\n  base <- strptime(dtList, \"%Y-%m-%d\")\n  bucket <- base + (roundedhrs * 60 * 60)\n  return (data.frame(table(bucket)))\n}\n\nfeaturesByTime <- function(ds, min1, min5, hr1, hr3, hr6) {\n  ds$min1_mean <- mean(min1)\n  ds$min1_max <- max(min1)\n  ds$min1_var <- safeVar(min1)\n  #ds$min1_skew <- skewness(min1, na.rm = FALSE, type = 1)\n  \n  ds$min5_mean <- mean(min5)\n  ds$min5_max <- max(min5)\n  ds$min5_var <- safeVar(min5)\n  #ds$min5_skew <- skewness(min5, na.rm = FALSE, type = 1)\n  \n  ds$hr1_mean <- mean(hr1)\n  ds$hr1_max <- max(hr1)\n  ds$hr1_var <- safeVar(hr1)\n  #ds$hr1_skew <- skewness(hr1, na.rm = FALSE, type = 1)\n  \n  ds$hr3_mean <- mean(hr3)\n  ds$hr3_max <- max(hr3)\n  ds$hr3_var <- safeVar(hr3)\n  #ds$hr3_skew <- skewness(hr3, na.rm = FALSE, type = 1)\n  \n  ds$hr6_mean <- mean(hr6)\n  ds$hr6_max <- max(hr6)\n  ds$hr6_var <- safeVar(hr6)\n  #ds$hr6_skew <- skewness(hr6, na.rm = FALSE, type = 1)\n  \n  return (ds)\n}\n##################################################################################################\n\n## IP FEATURE.\n## Compute the required features for the dataset\n## grouping it by the same IP\n\n## add new IP basedfeature columns to the given dataset\n## and add computed values to those columns\n## note that, the feature value for each of these columns will be\n## same for the provided dataset\nfeaturesByIP <- function(ds) {\n  ipClickFreq <- data.frame(table(ds$ip))\n  \n  ds$maxClicksIp <- max(ipClickFreq$Freq)\n  ds$noOfIPs <- nrow(ipClickFreq)\n  ds$clickToIpRatio <- nrow(ds)/ds$noOfIPs\n  ds$clickVarianceIp <- safeVar(ipClickFreq$Freq)\n  ds$entropyIp <- shannon.entropy(ipClickFreq$Freq)\n  \n  return (ds)\n}\n##################################################################################################\n\n## CHANNEL FEATURE.\nfeaturesByChannel <- function(ds) {\n  channelClickFreq <- data.frame(table(ds$channel))\n  \n  ds$maxClicksChannel <- max(channelClickFreq$Freq)\n  ds$noOfChannel <- nrow(channelClickFreq)\n  ds$clickToChannelRatio <- nrow(ds)/ds$noOfChannel\n  ds$clickVarianceChannel <- safeVar(channelClickFreq$Freq)\n  ds$entropyChannel <- shannon.entropy(channelClickFreq$Freq)\n  \n  return (ds)\n}\n##################################################################################################\n\n## DEVICE FEATURE.\nfeaturesByDevice <- function(ds) {\n  deviceClickFreq <- data.frame(table(ds$device))\n  \n  ds$maxClicksDevice <- max(deviceClickFreq$Freq)\n  ds$noOfDevices <- nrow(deviceClickFreq)\n  ds$clickToDeviceRatio <- nrow(ds)/ds$noOfDevices\n  ds$clickVarianceDevice <- safeVar(deviceClickFreq$Freq)\n  ds$entropyDevice <- shannon.entropy(deviceClickFreq$Freq)\n  \n  return (ds)\n}\n##################################################################################################\n\n#### utilities\nsafeVar <- function (x) {\n  return (ifelse(is.na(var(x)),0,var(x)))\n}\n\n## compute entropy\nshannon.entropy <- function(p) {\n  if (min(p) < 0 || sum(p) <= 0)\n    return(NA)\n  p.norm <- p[p>0]/sum(p)\n  -sum(log2(p.norm)*p.norm)\n}","execution_count":57,"outputs":[]},{"metadata":{"_uuid":"39ab8170be352808770ae5e39c7c433ac61737e5","trusted":true,"_cell_guid":"6d5c4eea-bdb2-48e2-9a98-1585527727ba"},"cell_type":"code","source":"##break the dataframe in smaller datasets -- group by app\ngroupedDS <- split(train, train$app)","execution_count":58,"outputs":[]},{"metadata":{"_uuid":"6f178900b1a8999fcd0efe3dc5be892ad3da1b51","trusted":true,"_cell_guid":"a070d902-7749-437d-b430-b671052212b5"},"cell_type":"code","source":"## iterate thru each dataset and add features to that part of the dataset\n## then append to a cumulative dataframe that will contain the dataframe with the added features\nfeaturesDF = c()\nfor(ds in groupedDS) {\n  min1 <- bucketInMins(ds$click_time, 1)\n  min5 <- bucketInMins(ds$click_time, 5)\n  min60 <- bucketInMins(ds$click_time, 60)\n  hr1 <- bucketInHrs(ds$click_time, 1)\n  hr3 <- bucketInHrs(ds$click_time, 3)\n  hr6 <- bucketInHrs(ds$click_time, 6)\n  \n  addedTimeFeat <- featuresByTime(ds, min1$Freq, min5$Freq, hr1$Freq, hr3$Freq, hr6$Freq)\n  addedIpFeat <- featuresByIP(addedTimeFeat)\n  addedDeviceFeat <- featuresByDevice(addedIpFeat)\n  addedChannelFeat <- featuresByChannel(addedDeviceFeat)\n  \n  featuresDF <- rbind(featuresDF, addedChannelFeat)\n  \n}","execution_count":59,"outputs":[]},{"metadata":{"_uuid":"c53e3790799fbfd072d9f47ef27513f429e933d6","trusted":true,"_cell_guid":"62ee4aee-7c43-4a2e-9bd5-bc81fd650a07"},"cell_type":"code","source":"df <- featuresDF","execution_count":60,"outputs":[]},{"metadata":{"scrolled":true,"_uuid":"b2fb03cae5639ca3dc2a27a5292bfa45a02a3956","trusted":true,"_cell_guid":"d31593c5-9873-4074-b196-6b0917749579"},"cell_type":"code","source":"dim(featuresDF)\nhead(featuresDF,3)","execution_count":61,"outputs":[]},{"metadata":{"_uuid":"faf23a9792026923cf35cf259a408f1e8ed5a366","trusted":true,"_cell_guid":"0a8b86c6-2d89-4334-9a7c-f21f6a80be65"},"cell_type":"code","source":"# removing the old features\nfeaturesDF <- featuresDF[,-1:-6]\nhead(featuresDF,2)\n","execution_count":62,"outputs":[]},{"metadata":{"_uuid":"841535cfdc50f4c3d00da886f4b614bc55a68b93","trusted":true,"_cell_guid":"80241014-303e-49ee-8717-443e611cba78"},"cell_type":"code","source":"dim(featuresDF)\n","execution_count":64,"outputs":[]},{"metadata":{"_uuid":"82b5cab743db1bb477218468a6d9e9362e4be868","trusted":true,"_cell_guid":"25fe15ff-e2bb-4653-bf7c-a485a894a616"},"cell_type":"code","source":"#Splitting the Data into 70% Training and 30% Testing\nindex<-sort(sample(nrow(featuresDF),round(.3*nrow(featuresDF))))\ntraining<-featuresDF[-index,]\ntest<-featuresDF[index,]","execution_count":66,"outputs":[]},{"metadata":{"_uuid":"aa71e695670bad5ddef07fc8aa0178f746055f98","trusted":true,"_cell_guid":"c7280329-ed69-4a33-95d5-7471d2f0740b"},"cell_type":"code","source":"dim(training)","execution_count":68,"outputs":[]},{"metadata":{"_uuid":"73507c40d67f9841bcd6c37670cbdab23bb370c9","trusted":true,"_cell_guid":"ccf5405c-6a66-428c-a8eb-7eb3c531557f"},"cell_type":"code","source":"dim(test)","execution_count":69,"outputs":[]},{"metadata":{"_uuid":"c46b201450fb3ea4e4bd62fddbfd98246742b676","_cell_guid":"f69350bd-512e-4f5f-a1aa-07ddbd562d87"},"cell_type":"markdown","source":"# Machine Learning Model on Unsampled Data"},{"metadata":{"_uuid":"2ab7960a9c79b9baf0967c52879536486baed11f","trusted":true,"_cell_guid":"698cd81d-a7f5-49a0-b6c3-58343e585d4f"},"cell_type":"code","source":"#Common function for Models\n#### utilities\n## factored fields prediction\nprediction <- function (algo, modelDS, testDf, col) { \n  predictField <- predict(modelDS,testDf[,-col])\n  predictionNonFactored(algo, predictField, testDf, col)\n}\n\n## non factored fields prediction\npredictionNonFactored <- function (algo, modelDS, testDf, col) { \n  table(predicted = modelDS, actual = testDf[,col])\n  accuracyRate <- sum(modelDS == testDf[,col])/nrow(testDf) * 100\n  errorRate <- sum(modelDS != testDf[,col])/nrow(testDf) * 100 \n  print(paste(algo, \": accuracyRate: \", accuracyRate, \"%, errorRate: \", errorRate, \"% \", sep=\"\"))\n}","execution_count":70,"outputs":[]},{"metadata":{"_uuid":"4319a7d2c30a1441665cd289605b3e578168dc7f","_cell_guid":"9c3d3b27-9ad7-433d-a250-0342212abd99"},"cell_type":"markdown","source":"# Naive Bayes - Baseline Model"},{"metadata":{"_uuid":"aaec38bee797584535d48d276b1008e4a571a90c","trusted":true,"_cell_guid":"4bd1146d-96f0-4e21-96f7-557045e39d04"},"cell_type":"code","source":"library(e1071)","execution_count":71,"outputs":[]},{"metadata":{"_uuid":"70d5aa85283d28dd465bce575bf53c77823e75fc","trusted":true,"_cell_guid":"18c81268-c6f2-40d2-818f-b61bef44f247"},"cell_type":"code","source":"nBayesAll <- naiveBayes(is_attributed ~., data=training)","execution_count":72,"outputs":[]},{"metadata":{"_uuid":"224c1474b9a9c4a5fedffd023ec2310beaeaface","trusted":true,"_cell_guid":"1ae74a70-215a-4c58-a703-6ea13a2f4a69"},"cell_type":"code","source":"## Naive Bayes classification using all variables \n\ncategory_all<-predict(nBayesAll,test)\n \n\ntable(NBayes_all=category_all,Predict=test$is_attributed)\n","execution_count":74,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e28ef65a4e57033eb572bde895c39ad8fdf90fc1"},"cell_type":"code","source":"NB_wrong<-sum(category_all!=test$is_attributed)\nNB_error_rate<-NB_wrong/length(category_all)\nNB_error_rate","execution_count":77,"outputs":[]},{"metadata":{"_uuid":"c6c44e59cfda76d1de3b3a6585144c764f4d1841"},"cell_type":"markdown","source":"So Using Naive Bayes, we are getting an Accuracy of 75% Approx. "},{"metadata":{"_uuid":"b00b5b5e0a47fc6ef2cecd14571b3537640f5998"},"cell_type":"markdown","source":"We are going to use this as a baseline model"},{"metadata":{"_uuid":"26b8f67e7c512e19106dcd9693ee1d62b3887335","_cell_guid":"9d19f22a-2b93-4623-ad59-9b14c0a3be7c"},"cell_type":"markdown","source":"# C4.5 Cart "},{"metadata":{"_uuid":"d0fcca43731ebd825804526d0ae4d740e93e4bb4","trusted":true,"_cell_guid":"6c86c3ff-8904-46ca-b0a3-e524d4af3910"},"cell_type":"code","source":"library(RWeka)\n#install.packages(\"rattle\")  # CART standard package\n#?install.packages()\n\n#install.packages(\"rpart\")\n#install.packages(\"rpart.plot\")     # Enhanced tree plots\n#install.packages(\"rattle\")         # Fancy tree plot\n#install.packages(\"RColorBrewer\")   # colors needed for rattle\nlibrary(rpart)\nlibrary(rpart.plot)  \t\t\t# Enhanced tree plots\n#library(rattle)           # Fancy tree plot\nlibrary(RColorBrewer)     # colors needed for rattle","execution_count":82,"outputs":[]},{"metadata":{"_uuid":"0b595a7150f69dd68a399919c0ed6c5be3224163","trusted":true,"_cell_guid":"bc70f3f2-ac74-48ff-9411-0e867e5c957d"},"cell_type":"code","source":"#Grow the tree \nCART_class<-rpart( is_attributed ~., data=training)\nrpart.plot(CART_class)\n","execution_count":83,"outputs":[]},{"metadata":{"_uuid":"6622504c24e441ae2c9fc4cc445efb93cc7b16b5","trusted":true,"_cell_guid":"108417ea-0cee-45bd-bb3e-d1dd348780bb"},"cell_type":"code","source":"#C4.5\nCART_predict2<-predict(CART_class,test, type=\"class\")\n","execution_count":89,"outputs":[]},{"metadata":{"_uuid":"36c48d9efa5205eae320ddde3762f55ccecf92f3","trusted":true,"_cell_guid":"914542bd-ae7d-4edc-b156-cfb3730dbc2e"},"cell_type":"code","source":"table(CART=CART_predict2,Predict=test$is_attributed)\nCART_wrong2<-sum(CART_predict2!=test$is_attributed)\nCART_error_rate2<-CART_wrong2/length(CART_predict2)\nCART_error_rate2 \n","execution_count":91,"outputs":[]},{"metadata":{"_uuid":"48b01704e70b7e436dad49c14b910bf68fe30bf6","trusted":false,"_cell_guid":"d4b7fa3e-1641-41ff-9d97-f8d6474fe795"},"cell_type":"markdown","source":"We are getting a pretty decent accuracy using C4.5"},{"metadata":{"_uuid":"fb768f5bdfea0fdf5c60b064958c0a16d60e1d9f"},"cell_type":"markdown","source":"Accuracy: 99.78%"},{"metadata":{"_uuid":"730a5f875060602224e3f7316eebd48dd40a32ac","_cell_guid":"068a50e7-75aa-49b7-ba66-11369e084012"},"cell_type":"markdown","source":"# Random Forest"},{"metadata":{"_uuid":"4421ca277d1747a5a4579343f147b13a8f3c31b3","trusted":true,"_cell_guid":"87126d07-2e75-4728-bb3b-971b16728bae"},"cell_type":"code","source":"rf <- randomForest(is_attributed~.,data=training,importance=TRUE,ntree=200)\nimportance(rf)\nvarImpPlot(rf)\n","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"b54da0fb7088b5820e5283afdeb11a5f54e8be47","trusted":true,"_cell_guid":"733fcabd-9538-4776-9179-44f07f411fa6"},"cell_type":"code","source":"prediction(\"Random Forest\", rf, test, 1)   #1 is col number of is_attributed","execution_count":1,"outputs":[]},{"metadata":{"_uuid":"df18811ab2bc7e85421cea6f3e2992e1d5058a85","trusted":false,"_cell_guid":"5ad37373-06bc-42da-b086-03f399d8e7a7"},"cell_type":"markdown","source":"# Data Resampling"},{"metadata":{"_uuid":"1d6b52b67597d62551297529df33f5271612aa3c","trusted":true,"_cell_guid":"70cfdbb8-6deb-4a87-93ce-8460d2ea5509"},"cell_type":"code","source":"head(training,2)","execution_count":2,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c216d9a9c9f9044b5be07ce4f8ea70758c57eb8d"},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"0912c8e28a8e5f83d9d1996107092a21e2c43e96"},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"606e8fefc5c4b7637573090ee529f8b8c792aeec"},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"de06acc9beccbc1809a08c387ae69f7422a09881"},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"8141ad2f006f4fd01a050f0da96159c8db47ffa1"},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"R","language":"R","name":"ir"},"language_info":{"mimetype":"text/x-r-source","name":"R","pygments_lexer":"r","version":"3.4.2","file_extension":".r","codemirror_mode":"r"}},"nbformat":4,"nbformat_minor":1}