# This R script will run on our backend. You can write arbitrary code here!

# Many standard libraries are already installed, such as randomForest
library(randomForest)

# The train and test data is stored in the ../input directory
train <- read.csv("../input/train.csv")
test  <- read.csv("../input/test.csv")

# We can inspect the train data. The results of this are printed in the log tab below
str(train)
str(test)

# What proportion of passengers survived (total, by sex, by class)?
prop.table(table(train$Survived))

survived_by_sex <- table(train$Sex,train$Survived)
prop.table(survived_by_sex,1)

prop.table(table(train$Pclass,train$Survived),1)

# Here we will plot the passenger survival by class and by sex
#train$Survived <- factor(train$Survived, levels=c(1,0))
#levels(train$Survived) <- c("Survived", "Died")
train$Pclass <- as.factor(train$Pclass)
levels(train$Pclass) <- c("1st Class", "2nd Class", "3rd Class")

png("1_survival_by_class.png", width=800, height=600)
mosaicplot(train$Pclass ~ train$Survived, main="Passenger Survival by Class",
           color=c("#8dd3c7", "#fb8072"), shade=FALSE,  xlab="", ylab="",
           off=c(0), cex.axis=1.4)
dev.off()

png("2_survival_by_sex.png")
plot(train$Survived ~ train$Sex, main= "Passenger Survival by Sex", xlab="", ylab="",
        col=c("#33EE55","#FF0044"))
dev.off()

png("3_survival_by_sex.png")
barplot(survived_by_sex, col="wheat")
dev.off()

# Histogram of the passenger survival according to their age
png("4_survival_by_age.png")
plot(train$Survived ~ train$Age)
dev.off()

# Passenger survival according to the fare they paid

# New variable to class the fares in  groups
train$fare_group <- "Fare1"
train$fare_group[train$Fare >=5 & train$Fare <10] <- "Fare2"
train$fare_group[train$Fare >=10 & train$Fare <15] <- "Fare3"
train$fare_group[train$Fare >=15 & train$Fare <20] <- "Fare4"
train$fare_group[train$Fare >=20 & train$Fare <30] <- "Fare5"
train$fare_group[train$Fare >=30 & train$Fare <100] <- "Fare6"
train$fare_group[train$Fare >=100] <- "Fare7"

# Histogram of the passenger survival according to the fare paid
png("5_survival_by_fare.png", width=800, height=600)
mosaicplot(train$fare_group ~ train$Survived, main="Passenger Survival by fare",
           color=c("#8dd3c7", "#fb8072"), shade=FALSE,  xlab="", ylab="",
           off=c(0), cex.axis=1.4)
dev.off()

# Aggregate with subsets according to the fare group and sex
test1 <- aggregate(Survived ~ fare_group + Sex + Pclass, data = train, FUN= function(x) {sum(x)/length(x)})
# with test 1 we can see that female in 3rd class in fare group 3, 5 or 6 are likely to die

# Prediction with test data: all passengers died (Survived == 0)
test$Survived <- rep(0,418)

# Prediction: all women survived and all men died
test$Survived[test$Sex == "female"] <- 1

# Prediction: women in 3rd class and in fare group 3, 5 or 6 died
test$Survived[test$Sex == "female" & test$Pclass == 3 & test$Fare >= 10 & test$Fare <100] <- 0
test$Survived[test$Sex == "female" & test$Pclass == 3 & test$Fare >= 15 & test$Fare <20] <- 1

# Create the csv file for submission
submission <- data.frame(PassengerId = test$PassengerId, Survived = test$Survived)
write.csv(submission, file="titanic_submission.csv", row.names=FALSE)







