---
title: 'Fraud Detection by Random Forest,DT and SVM'
author: 'Swamy S M'
date: '21st Mar 2018'

---
  
# Introduction
  
  This is again one of the classification problem

China's largest independent big data service platform challenged us to build an algorithm that predicts whether a user will download an app after clicking a mobile app ad

And the data given to us is masked data (we only can see the codes not the actual data)

This is one of the practise everyone need to adopt as per GDPR initiative (General Data Protection Regulation), this must be adopted for all company which is processing US data

Just for information about GDPR, GDPR will take effect on May 25th, 2018 for all European data processing countries and Non-compliance to GDPR will have heavy penalties associated with it.


#Data loading and consolidation

---

```{r, message = FALSE}
## Let’s load all the required libraries
library(lubridate)
library(caret)
library(dplyr)
library(DMwR)
library(ROSE)
library(ggplot2)
library(randomForest)
library(rpart)
library(rpart.plot)
library(data.table)
library(e1071)
library(gridExtra)

###Read the data and check the structure of both train and test

train <-fread('../input/train_sample.csv', stringsAsFactors = FALSE, data.table = FALSE)
test <-fread('../input/test.csv', stringsAsFactors = FALSE, data.table = FALSE)

str(train)
str(test)

## There is no difference between train and test data except we need to predict target (is_attributed) in test 
## and attributed_time (Time taken to download Application) is not given in test data)


```

#Missing value checking and estimation

```{r, message = FALSE}
colSums(is.na(train))

##There is no missing value at all, data is very clean and clear

colSums(train=='')

##Attributes_time (Time taken to download) having blank entries, this is logically correct 

##Lets check the target variable how many are not downloaded in train data
table(train$is_attributed)

##Our assumption is correct since blank entries in Attributes_time is matching with Application not downlaoded in train data.
##As it’s logically correct, we don't need do any further action on this
##And also notice that, this variable is not present in test data, so no point of keeping it in the train data too

train$attributed_time=NULL
```

# Feature engineering
```{r, message = FALSE}

####As most of the feature variable is masked and we don’t have much scope in feature engineering
####However, we can extract some of variables out of click_time to use that in our future analysis 

##Convert click_time into proper date and time format
train$click_time<-as.POSIXct(train$click_time,
                             format = "%Y-%m-%d %H:%M",tz = "America/New_York")

##Get the data
train$year=year(train$click_time)
train$month=month(train$click_time)
train$days=weekdays(train$click_time)
train$hour=hour(train$click_time)

##After getting new feature, let’s remove original "click_time" variable

train$click_time=NULL

##Check the unique number for each of the feauter
apply(train,2, function(x) length(unique(x)))

##By looking into unique value, we can see data collected for one month in a year, so no point of keeping month and year variables

train$month=NULL
train$year=NULL

##Convert variables into respective data type

train$is_attributed=as.factor(train$is_attributed)
train$days=as.factor(train$days)
```

#Exploratory data analysis and checking importance of feature

```{r, message = FALSE}

###App was downloaded v/s App id for marketing

p1=ggplot(train,aes(x=is_attributed,y=app,fill=is_attributed))+
  geom_boxplot()+
  ggtitle("Application ID v/s Is_attributed")+
  xlab("App ID") +
  labs(fill = "is_attributed")  

p2=ggplot(train,aes(x=app,fill=is_attributed))+
  geom_density()+facet_grid(is_attributed~.)+
  scale_x_continuous(breaks = c(0,50,100,200,300,400))+
  ggtitle("Application ID v/s Is_attributed")+
  xlab("App ID") +
  labs(fill = "is_attributed")  


p3=ggplot(train,aes(x=is_attributed,y=app,fill=is_attributed))+
  geom_violin()+
  ggtitle("Application ID v/s Is_attributed")+
  xlab("App ID") +
  labs(fill = "is_attributed")  


grid.arrange(p1,p2, p3, nrow=2,ncol=2)

##Observe Different pattern and shape in all the graph of App was downloaded v/s App id in the market, especially clear differentiation in Boxplot
##This is definitely going to be one of the important feature to differentiate user downloaded the application or not
```

```{r, message = FALSE}

###App was downloaded vs OS version id of user mobile phone
p4=ggplot(train,aes(x=is_attributed,y=os,fill=is_attributed))+
  geom_boxplot()+
  ggtitle("Os version v/s Is_attributed")+
  xlab("OS version") +
  labs(fill = "is_attributed")  


p5=ggplot(train,aes(x=os,fill=is_attributed))+
  geom_density()+facet_grid(is_attributed~.)+
  scale_x_continuous(breaks = c(0,50,100,200,300,400))+
  ggtitle("Os version v/s Is_attributed ")+
  xlab("Os version") +
  labs(fill = "is_attributed")


p6=ggplot(train,aes(x=is_attributed,y=os,fill=is_attributed))+
  geom_violin()+
  ggtitle("Os version v/s Is_attributed")+
  xlab("Os version") +
  labs(fill = "is_attributed")  


grid.arrange(p4,p5, p6, nrow=2,ncol=2)

####No such differentiation exist between 2 classes, this is definitely not an important feature for prediction


```

```{r, message = FALSE}

###App was downloaded v/s ip address of click.
p7=ggplot(train,aes(x=is_attributed,y=ip,fill=is_attributed))+
  geom_boxplot()+
  ggtitle("IP Address v/s Is_attributed")+
  xlab("Ip Adresss of click") +
  labs(fill = "is_attributed")  


p8=ggplot(train,aes(x=ip,fill=is_attributed))+
  geom_density()+facet_grid(is_attributed~.)+
  scale_x_continuous(breaks = c(0,50,100,200,300,400))+
  ggtitle("IP Address v/s Is_attributed")+
  xlab("Ip Adresss of click") +
  labs(fill = "is_attributed")  



p9=ggplot(train,aes(x=is_attributed,y=ip,fill=is_attributed))+
  geom_violin()+
  ggtitle("IP Address v/s Is_attributed")+
  xlab("Ip Adresss of click") +
  labs(fill = "is_attributed")  

grid.arrange(p7,p8, p9, nrow=2,ncol=2)


####IP Address definitely could play very important role in prediction as clear diffrenation exist between 2 groups

```  

```{r, message = FALSE}

###App was downloaded v/s device type id of user mobile phone

p10=ggplot(train,aes(x=device,fill=is_attributed))+
  geom_density()+facet_grid(is_attributed~.)+
  ggtitle("Device type v/s Is_attributed")+
  xlab("Device Type ID") +
  labs(fill = "is_attributed")  


p11=ggplot(train,aes(x=is_attributed,y=device,fill=is_attributed))+
  geom_boxplot()+
  ggtitle("Device type v/s Is_attributed")+
  xlab("Device Type ID") +
  labs(fill = "is_attributed")  


p12=ggplot(train,aes(x=is_attributed,y=device,fill=is_attributed))+
  geom_violin()+
  ggtitle("Device type v/s Is_attributed")+
  xlab("Device Type ID") +
  labs(fill = "is_attributed")  

grid.arrange(p10,p11, p12, nrow=2,ncol=2)

##No such differentiation exist between the group, not an important variable for our analysis


```

```{r, message = FALSE}


###App was downloaded v/s channel id of mobile ad publisher

p13=ggplot(train,aes(x=channel,fill=is_attributed))+
  geom_density()+facet_grid(is_attributed~.)+
  ggtitle("Channel v/s Is_attributed")+
  xlab("Channel of mobile") +
  labs(fill = "is_attributed")  


p14=ggplot(train,aes(x=is_attributed,y=channel,fill=is_attributed))+
  geom_boxplot()+
  ggtitle("Channel v/s Is_attributed")+
  xlab("Channel of mobile") +
  labs(fill = "is_attributed")  

p15=ggplot(train,aes(x=is_attributed,y=channel,fill=is_attributed))+
  geom_violin()+
  ggtitle("Channel v/s Is_attributed")+
  xlab("Channel of mobile") +
  labs(fill = "is_attributed")  

grid.arrange(p13,p14, p15, nrow=2,ncol=2)

###Channel definitely got some predictive power, definitely we can use this for our feature analysis
```


```{r, message = FALSE}

###Does specific hour play any role in downloading 

p16=ggplot(train,aes(x=hour,fill=is_attributed))+
  geom_density()+facet_grid(is_attributed~.)+
  ggtitle("Hour v/s Is_attributed ")+
  xlab("Hour") +
  labs(fill = "is_attributed")  

p17=ggplot(train,aes(x=is_attributed,y=hour,fill=is_attributed))+
  geom_boxplot()+
  ggtitle("Hour v/s Is_attributed")+
  xlab("Hour") +
  labs(fill = "is_attributed")  

p18=ggplot(train,aes(x=is_attributed,y=channel,fill=is_attributed))+
  geom_violin()+
  ggtitle("Hour v/s Is_attributed")+
  xlab("Hour") +
  labs(fill = "is_attributed")  

grid.arrange(p16,p17, p18, nrow=2,ncol=2)


##There is slight difference in both the distribution, we can say least important feature

```

```{r, message = FALSE}

###Does Particular day play any role in downloading application?

p19=ggplot(train,aes(x=days,fill=is_attributed))+
  geom_density()+facet_grid(is_attributed~.)+
  ggtitle("Day of a week v/s Is_attributed ")+
  xlab("Os version") +
  labs(fill = "is_attributed")  


p20=ggplot(train,aes(x=days,fill=is_attributed))+geom_density(col=NA,alpha=0.35)+
  ggtitle("days v/s click")+
  xlab("Day of a week v/s Is_attributed ") +
  ylab("Total Count") +
  labs(fill = "is_attributed")  

grid.arrange(p19,p20, ncol=2)

###Seems nothing in day variable

```

##Validating feature insights arrived

```{r, message = FALSE}

##Lets randomly apply any model 
# 1) for all the feature
# 2) for selected feature through exploratory data analysis


# 1) for all the feature
set.seed(1234)
cv.10 <- createMultiFolds(train$is_attributed, k = 10, times = 10)


ctrl <- trainControl(method = "repeatedcv", number = 10, repeats = 10,
                     index = cv.10)

set.seed(1234)

Model_CDT <- train(x = train[,-6], y = train[,6], method = "rpart", tuneLength = 30,
                   trControl = ctrl)

PRE_VDTS=predict(Model_CDT$finalModel,data=train,type="class")
confusionMatrix(PRE_VDTS,train$is_attributed)

##Check the accuracy, Even though overall accurcay is very high but specificity is too low

# 2 ) For selcted feauter
train$days=NULL
train$os=NULL
train$device=NULL

set.seed(1234)

Model_CDT1 <- train(x = train[,-4], y = train[,4], method = "rpart", tuneLength = 30,
                    trControl = ctrl)

PRE_VDTS1=predict(Model_CDT1$finalModel,data=train,type="class")
confusionMatrix(PRE_VDTS1,train$is_attributed)

###Second model gives the same accuracy, however there is drastic change in specificity 
###So I will start using only selected feature for our rest of modeling course
```

#Data partition
```{r, message = FALSE}
##Before doing anything, Lets divide the data into training and testing data using caret package

set.seed(5000)
ind=createDataPartition(train$is_attributed,times=1,p=0.7,list=FALSE)
train_val=train[ind,]
test_val=train[-ind,]

####Check the proprtion  and its the same
round(prop.table(table(train$is_attributed)*100),digits = 3)
round(prop.table(table(train_val$is_attributed)*100),digits = 3)
round(prop.table(table(test_val$is_attributed)*100),digits = 3)

##Notice, how well caret divided the data into 70% to 30% ratio and also it make sure that no change in the proportion of target variable

```

##Data Balancing using Smote

```{r, message = FALSE}
###In the backend, I have checked all the techniques such as up sampling, down sampling, Rose and smote, among them Smote come out with good accuracy

##Lets apply smote and try to balance the data
set.seed(1234)
smote_train = SMOTE(is_attributed ~ ., data  = train_val)                         
table(smote_train$is_attributed) 
```


##Machine learning algorithms and cross validation 

###Decision tree
```{r, message = FALSE}

set.seed(1234)
cv.10 <- createMultiFolds(smote_train$is_attributed, k = 10, times = 10)

# Control
ctrl <- trainControl(method = "repeatedcv", number = 10, repeats = 10,
                     index = cv.10)
set.seed(1234)
##Train the data
Model_CDT <- train(x = smote_train[,-4], y = smote_train[,4], method = "rpart", tuneLength = 30,
                   trControl = ctrl)

rpart.plot(Model_CDT$finalModel,extra =  3,fallen.leaves = T)


PRE_VDTS=predict(Model_CDT$finalModel,newdata=test_val,type="class")
confusionMatrix(PRE_VDTS,test_val$is_attributed)

##We are able to complete decision tree with 0.94% accuracy, and specificity increased to 0.78% (Remember, drastic increase in specificty after data balance)
```

###Random forest

```{r, message = FALSE}
cv.10 <- createMultiFolds(smote_train$is_attributed, k = 10, times = 10)

# Control
ctrl <- trainControl(method = "repeatedcv", number = 10, repeats = 10,
                     index = cv.10)
set.seed(1234)
set.seed(1234)
rf.5<- train(x = smote_train[,-4], y = smote_train[,4], method = "rf", tuneLength = 3,
             ntree = 100, trControl =ctrl)

rf.5

pr.rf=predict(rf.5,newdata = test_val)

confusionMatrix(pr.rf,test_val$is_attributed)

##Random forest model giving us 95% accuracy, 1% better than decision tree but notice there is no much change specificity 
```
###Support Vector Machine
###Linear Support vector Machine

```{r, message = FALSE}

##Before going into model,  lets tune the cost Parameter

set.seed(1234)
liner.tune=tune.svm(is_attributed~.,data=smote_train,kernel="linear",cost=c(0.1,0.5,1,5,10,50))

liner.tune
###Lets get a best.liner model  
best.linear=liner.tune$best.model

##Predict data

best.test=predict(best.linear,newdata=test_val,type="class")
confusionMatrix(best.test,test_val$is_attributed)

##Accuracy going down in Linear SVM, SVM is not a good model for this data
```
###Radial Support vector Machine

```{r, message = FALSE}


######Lets go to non liner SVM, Radial Kerenl

set.seed(1234)
rd.poly=tune.svm(is_attributed~.,data=smote_train,kernel="radial",gamma=seq(0.1,5))

summary(rd.poly)

best.rd=rd.poly$best.model


##Lets Predict test data
pre.rd=predict(best.rd,newdata = test_val)

confusionMatrix(pre.rd,test_val$is_attributed)
##Even though Radial kernel doing better than linear, at Overall level accuracy is not good
```

Conclusion: We could have achieved 99% accuracy by simply using given data without making class balance.
At the end of the day, we need to focus on genuine user as well, if we are telling all user to fraudulent then no point in making any model.
This is it Kaggle community, Much appreciated for any feedback, suggestion


