# %% [code]
# %% [code]
---
title: "Predicting survival in the sinking of the Titanic: Who died?"
author: "André Pardal"
date: "6/22/2022"
output: html_document
editor_options: 
  chunk_output_type: console
---

```{r Loading libraries and data}

## EN: Loading libraries ##
## PT: Carregando as bibliotecas ##
library(ggplot2)
library(dplyr)
library(randomForest)

## EN: Loading data ##
## PT: Carregando os dados ##
train <- read.csv('../input/titanic/train.csv', stringsAsFactors = F)
test  <- read.csv('../input/titanic/test.csv',  stringsAsFactors = F)

## EN: Merging dataframes ##
## PT: Juntando os conjuntos de dados ##
full_df <- dplyr::bind_rows(train, test)
str(full_df)
summary(full_df)

## EN: Remove dataframes 'test' and 'train' (We'll split them back after) ##
## PT: Removendo os dataframes 'test' e 'train' (Dividiremos os dados neles de novo depois) ##
rm(test, train)

```

```{r Dealing with missing data}

## EN: Check NAs ##
## PT: Verificando dados faltantes ##

colSums(is.na(full_df)) 
## EN: There's a lot of missing data in 'Age' (263) ##
## PT: Há muitos dados faltantes em 'Age' (263) ##

## EN: There's 1 NA in 'Fare' ##
## PT: Há 1 dado faltante em 'Fare' ##

colSums(full_df == "")
## EN: There is 2 missing data for 'Embarked' ##
## PT: Há 2 dados faltantes em 'Embarked' ##

## EN: Aggregating 'Fare' per 'Embarked' as mean ##
## PT: Agregando 'Fare' pelos níveis de 'Embarked' como média ##
Fare_av <- aggregate(Fare ~ Embarked, full_df, mean)
Fare_av 
## EN: Both paid Fare = 80 ##
## PT: Ambos pagaram Fare = 80 ##

## EN: These passengers most likely embarked from 'C' given the paid Fare (80) ##
## EN: So let's replace missing data for "C" ##
full_df$Embarked[full_df$Embarked == ""] <- "C"

## EN: Now, let's consider that the unique missing 'Fare' is the mean Fare from those whom embarked from 'S' ##
## PT: Agora, vamos considerar que o único dado faltante de 'Fare' é a média dos valores pagos por quem embarcou em 'S' ##
full_df$Fare[is.na(full_df$Fare)] <- mean(full_df$Fare[full_df$Embarked == "S"], na.rm = T)
####


## EN: Dealing with Age: Let's use people's title to estimate their ##
## PT: Lidando com 'Agea: Vamos usar o título das pessoas para estimar a idade delas ##

## EN: Getting 'Title' from 'Name' ##
## PT: Obtendo 'Title' da coluna 'Name' ##
full_df$Title <- gsub('(.*, )|(\\..*)', '', full_df$Name)
summary(as.factor(full_df$Title))

## EN: Let's group common/similar titles ##
## PT: Vamos agrupar títulos comuns/similares ##
full_df$Title[full_df$Title %in% c('Mlle', 'Lady', 'Dona', 'Ms')]                      <- 'Miss' 
full_df$Title[full_df$Title %in% c("the Countess", "Mme")]                             <- 'Mrs' 
full_df$Title[full_df$Title %in% c('Capt','Col','Don','Jonkheer','Major','Rev','Sir')] <- 'Mr' 

full_df$Title<- as.factor(full_df$Title)
summary(as.factor(full_df$Title))

## EN: Calculating the mean per title
## First:
aggregate(full_df[,6], by = list(full_df$Title), FUN = function(x) {sum(is.na(x))})
aggregate(full_df[,6], by = list(full_df$Title), FUN = function(x) {length((x))})

age_mean <- aggregate(Age ~ Title, full_df, mean)
age_mean
## EN: These 'Age' means shall be replaced to where there's NA (but per 'Title') ##

## EN: Replacing NAs with the mean 'Age' per 'Title' ##
## PT: Transformando NAs em médias de 'Age' por 'Title' ##
full_df$Age[full_df$Title=='Dr' &     is.na(full_df$Age)] <- mean(full_df$Age[full_df$Title == 'Dr'],     na.rm = T)
full_df$Age[full_df$Title=='Master' & is.na(full_df$Age)] <- mean(full_df$Age[full_df$Title == 'Master'], na.rm = T)
full_df$Age[full_df$Title=='Miss' &   is.na(full_df$Age)] <- mean(full_df$Age[full_df$Title == 'Miss'],   na.rm = T)
full_df$Age[full_df$Title=='Mrs' &    is.na(full_df$Age)] <- mean(full_df$Age[full_df$Title == 'Mrs'],    na.rm = T)
full_df$Age[full_df$Title=='Mr' &     is.na(full_df$Age)] <- mean(full_df$Age[full_df$Title == 'Mr'],     na.rm = T)

summary(full_df$Age)
## EN: No more NAs in Age ##
## PT: Não há mais NAs em 'Age' ##

rm(Fare_av, age_mean)

```

```{r Exploring data}

## EN: ## Verifying dataframe structure and fixing variables ##
## PT: ## Verificando a estrutura do conjunto de dados e consertando as variáveis ##
str(full_df)

full_df$Survived    <- as.factor(full_df$Survived)
full_df$Sex         <- as.factor(full_df$Sex)
full_df$Embarked    <- as.factor(full_df$Embarked)
full_df$PassengerId <- as.factor(full_df$PassengerId)
full_df$Pclass      <- as.factor(full_df$Pclass)

str(full_df)
summary(full_df)

## EN: Plotting for exploration ##
## PT: Plotando figuras para explorar padrões ##

full_df[1:891,] %>% ## Categorical predictors ##
  dplyr::select(-PassengerId,-Name, -Ticket, -Cabin, -Age, -Fare) %>% 
  tidyr::gather(key='fatores', value='valores', -Survived) %>% 
  ggplot(aes(x = valores, 
             fill = Survived)) + 
  geom_bar(alpha =.8, position= "fill")+ 
  scale_fill_manual(values=c("#69b3a2", "#404080"))+
  facet_wrap(~ fatores, ncol=3, scales = 'free') +
  labs(x = 'Categories', y = 'Count') +
  theme_classic()

full_df[1:891,] %>% ## Continuous predictors ##
  dplyr::select(Survived, Age, Fare) %>% 
  tidyr::gather(key='fatores', value='valores', -Survived) %>% 
  ggplot(aes(x = valores, 
             fill = Survived)) + 
  geom_histogram(alpha =.6, position = 'identity') + 
  scale_fill_manual(values=c("#69b3a2", "#404080"))+
  facet_wrap(~ fatores, ncol=3, scales = 'free') +
  labs(x = 'Values', y = 'Count') +
  theme_classic()

## EN: Create new variable Family size given that SibSp and Parch have a similar behavior ##
## PT: Criando uma nova variável 'FamilySize' ##
full_df$FamilySize <- full_df$SibSp + full_df$Parch + 1;

## EN: Plotting ##
## PT: Visualizando ##
  ggplot(full_df[1:891,], 
         aes(x = as.numeric(FamilySize), fill = Survived)) + 
  geom_bar(alpha = .8, position = "fill")+ 
  scale_fill_manual(values=c("#69b3a2", "#404080"))+
  labs(x = 'Family size (no. of people)', y = 'Count') +
  theme_classic()

```

```{r Modelling using ML - Random Forests}

## EN: Fitting a random forest model ##
## PT: Ajustando um modelo 'random forest' ##

## EN: First, let's split full_df back to 'train' ##
## PT: Primeiro, vamos dividir o 'full_df' em 'train' de novo ##
trainDF <- full_df[1:891,]
summary(trainDF)

## EN: Index for splitting trainDF (80-20) ##
splitIndex <- sample(round(nrow(trainDF)))

## EN: Get 80% of the dataframe 'trainDF' as 'titanicTrain' and 20% as 'titanicTest'##
titanicTrain <- trainDF[1:round(length(splitIndex)*0.8),]
titanicTest  <- trainDF[(round(length(splitIndex)*0.8)+1):nrow(trainDF),]

## EN: Fitting a Machine Learning Random Forests using 'titanicTrain' ##
model1 <- caret::train(Survived ~ Age + Pclass + Sex + Title +
                                  FamilySize + Fare + Embarked, 
                       titanicTrain, 
                       method = "ranger",
                       tuneLength = 6,
                       seed = 135, ## For reproducibility! ##
                       trControl = caret::trainControl(method = "repeatedcv",
                                                       number = 10,
                                                       repeats = 10,
                                                       verboseIter = F))
print(model1)

## Getting predictions for df 'titanicTest' based on 'model1' ##
titanicTest$pred <- predict(model1, titanicTest, type = "raw") 

## Comparing 'predicted vs observed' survival ##
confusionMatrix(titanicTest$pred, titanicTest$Survived) 
## Accuracy 0.8371: pretty good ##

rm(titanicTrain, titanicTest)


## EN: Now, let's train the model with full train data (i.e., trainDF) ##
## PT: Agora, vamos treinar o modelo com o conjunto de treinamento completo (i.e., trainDF) ##
model2 <- caret::train(Survived ~ Age + Pclass + Sex + Title +
                                  FamilySize + Fare + Embarked, 
                       trainDF, ## Whole training dataset ##
                       method = "ranger",
                       tuneLength = 6,
                       seed = 139,
                       trControl = caret::trainControl(method = "repeatedcv",
                                                       number = 10,
                                                       repeats = 10,
                                                       verboseIter = F))
print(model2)

## EN: Let's get the 'test' dataset again ##
## PT: Vamos obter o datagrame 'test' novamente ##
testDF <- full_df[892:1309,]
summary(testDF)

## EN: Predicted survival (0,1) for test dataset ##
## PT: Predizendo a sobrevivência (0 ou 1) para o conjunto de dados 'test' ##
testDF$Survived <- predict(model2, testDF, type = "raw")

## EN: Getting the final solution ##
## PT: Obtendo a solução final ##
solution <- data.frame(PassengerID = testDF$PassengerId, Survived = testDF$Survived)

## EN: Export .csv file with predictions ##
## PT: Exportando o arquivo .csv com as predições ##
readr::write_csv(x = solution, file = "submission.csv")

## The end! ##
## Fim ##
```
