
#My_RanDom_Forest_prediction_test

# Many standard libraries are already installed, such as randomForest
library(randomForest)

# The train and test data is stored in the ../input directory
train <- read.csv("../input/train.csv")
test  <- read.csv("../input/test.csv")

library(dplyr)
library(rpart)

set.seed(111)

#merge into a dataframe
all_data <- bind_rows(train,test)
all_data<-as.data.frame(all_data,na.rm=TRUE)
str(all_data) #Used for finding the all the information
names(all_data)#Used for finding the header names

#we have missed some data in Age 
is.na(all_data$Fare[1044])#cheaking for missing data, as was given in assignment 
#this is TRUE, passenger on row 1044 has NA. We can replace it with the median Fare value.
all_data$Fare[1044]<- median(all_data$Fare, na.rm=TRUE)
is.na(all_data$Embarked[c(62,830)])#cheaking for missing data, as was given in assignment 
all_data$Embarked[c(62,830)]<-"S"# we give them "S", relying on the majority
all_data$Embarked<-factor(all_data$Embarked)#very important step to factorize embarkment codes (if you factorize it,
#it willwork enem with missing factors, which were letters "S", "N", e.i. characters )

all_data$Embarked<-factor(all_data$Embarked)
#to fill missing data for Age we use prediction method as decision-tree with method "anova" for continious variable
library(rpart)
prediction_age <- rpart(Age~Pclass+Sex+SibSp+Parch+Fare+Embarked, data=all_data[!is.na(all_data$Age),], 
                        method="anova")
all_data$Age[is.na(all_data$Age)]<-predict(prediction_age, all_data[is.na(all_data$Age),])

#split data again to train and test
train_again<-all_data[1:891,]
test_again<-all_data[892:1309,]
#don't forget to factorize Survived 
#as.factor(Survived)~Pcalss+Sex+Age
set.seed(112)
my_forest<-randomForest(as.factor(Survived)~Pclass+Sex+Age+SibSp+Parch+Fare+Embarked,
                        data=train_again, importance=TRUE, ntree=1000)
my_prediction<- predict(my_forest, newdata = test_again)
varImpPlot(my_forest)
#create a data frame with two columns PassengerID&Survived
my_solution<-data.frame(PassengerId=test_again$PassengerId, Survived = my_prediction)
write.csv(my_solution, file="my_RF_solution.csv", row.names = FALSE)