# %% [code]
# %% [code]
---
title: 'Spaceship Titanic: Who was transported?'
author: 'André Pardal'
date: '`r Sys.Date()`'
output:
  html_document:
    number_sections: true
    toc: true
---


```{r Loading libraries and data}

## Loading libraries ##
library(tidyverse)
library(caret)

## Loading data ##
train <- read.csv(file = "../input/spaceship-titanic/train.csv", stringsAsFactors = F)
test  <- read.csv(file = "../input/spaceship-titanic/test.csv",  stringsAsFactors = F)

## Merging dataframes ##
full_df <- dplyr::bind_rows(train, test)

## Replace empty character cells to 'NA' ##
full_df[full_df ==""] <- NA
glimpse(full_df)

## Remove dataframes 'test' and 'train' (We'll put them back together ##
rm(test, train)
```


```{r Dealing with missing data}

## Check NAs ##
colSums(is.na(full_df)) 

## Split Cabin columns into deck, num and side columns
full_df <- separate(full_df, Cabin, c('deck', 'num', 'side'))

# Convert datatype of numeric or categorical data respectively 
full_df$num <- as.numeric(full_df$num)
full_df     <- as.data.frame(unclass(full_df), stringsAsFactors = T)

## Recode the categorical data 
full_df2 <- full_df %>% 
            mutate(CryoSleep =   factor(case_when(CryoSleep ==   'False' ~ 0, CryoSleep  == 'True' ~ 1)),
                   Transported = factor(case_when(Transported == 'False' ~ 0,Transported == 'True' ~ 1)),
                   VIP =         factor(case_when(VIP ==         'False' ~ 0,        VIP == 'True' ~ 1))) %>%
            select(-c(PassengerId, Name, Transported)) %>% # Ignore columns we won't impute data #
            mice::mice(method = "rf")    %>% # Use mice to impute data #
  complete()

## Check the number of missing value after imputation 
colSums(is.na(full_df2)) ## No more NAs ##

## Finishing organizing dataframe ##
full_df2 <- full_df2 %>% 
            mutate(PassengerId = full_df$PassengerId, ## get 'PassengerId' back from original data ##
                   Name = full_df$Name,               ## get 'Name' back from original data ##
                   Transported = full_df$Transported, ## get 'Transported' back from original data ##
                   TotalSpent = rowSums(full_df2[c(9:13)])) ## create new column with total spent value ## 

colSums(is.na(full_df2)) 
## Only Name and Transported with NAs: OK ##

rm(full_df)
```


```{r Exploring data}

## Plotting for exploration ##

full_df2[1:8693,] %>% ## Categorical predictors ##
  dplyr::select(Transported, CryoSleep, deck, side, VIP, Destination, HomePlanet) %>% 
  tidyr::gather(key='fatores', value='valores', -Transported) %>% 
  ggplot(aes(x = valores, 
             fill = Transported)) + 
  geom_bar(alpha =.8, position= "fill")+ 
  scale_fill_manual(values=c("#69b3a2", "#404080"))+
  facet_wrap(~ fatores, ncol=3, scales = 'free') +
  labs(x = 'Categories', y = 'Proportion') +
  theme_classic()

full_df2[1:8693,] %>% ## Continuous predictors ##
  dplyr::select(Transported, TotalSpent, Age, num, RoomService, FoodCourt, ShoppingMall, Spa, VRDeck) %>% 
  tidyr::gather(key='fatores', value='valores', -Transported) %>% 
  ggplot(aes(x = valores, 
             fill = Transported)) + 
  geom_histogram(alpha =.5, position = 'identity') + 
  scale_fill_manual(values=c("#69b3a2", "#404080"))+
  facet_wrap(~ fatores, ncol=3, scales = 'free') +
  labs(x = 'Values', y = 'Count') +
  theme_classic()


full_df2[1:8693,] %>% 
   select(TotalSpent, RoomService, FoodCourt, ShoppingMall, Spa, Age, VRDeck) %>% 
     PerformanceAnalytics::chart.Correlation(histogram = T, pch = 19, method = c("pearson"))

```


```{r Modelling using ML - Random Forests}
## Fitting a random forest model ##

## First, let's split full_df2 back to 'train' ##
trainDF <- full_df2[1:8693,]

## Recoding 'Transported' to 0 or 1 ##
trainDF <- trainDF %>% 
            mutate(Transported = factor(case_when(Transported == 'False' ~ 0,Transported == 'True' ~ 1)))

## Index for splitting trainDF (75-25) ##
splitIndex <- sample(round(nrow(trainDF)))

## Get 75% of the dataframe 'trainDF' as 'spaceTrain' and 20% as 'spaceTest'##
spaceTrain <- trainDF[1:round(length(splitIndex)*0.75),]
spaceTest  <- trainDF[(round(length(splitIndex)* 0.75)+1):nrow(trainDF),]

## Fitting a Machine Learning Random Forests using 'spaceTrain' ##
model1 <- caret::train(Transported ~ CryoSleep + deck + side + VIP + Destination +
                                     HomePlanet + TotalSpent + Age + num + RoomService +
                                     FoodCourt + ShoppingMall + Spa + VRDeck,
                       spaceTrain, 
                       method = "ranger",
                       #tuneLength = 6,
                       seed = 133, ## For reproducibility! ##
                       trControl = caret::trainControl(method = "repeatedcv",
                                                       number = 5,
                                                       repeats = 5,
                                                       verboseIter = T))
print(model1)

## Getting predictions for df 'spaceTest' based on 'model1' ##
spaceTest$pred <- predict(model1, spaceTest, type = "raw") 

## Comparing 'predicted vs observed' survival ##
confusionMatrix(spaceTest$pred, spaceTest$Transported) 
## Accuracy 0.7879: pretty good ##

rm(spaceTrain, spaceTest)

## Now, let's train the model with full train data (i.e., trainDF) ##
model2 <- caret::train(Transported ~ CryoSleep + deck + side + VIP + Destination +
                                     HomePlanet + TotalSpent + Age + num + RoomService +
                                     FoodCourt + ShoppingMall + Spa + VRDeck,
                       trainDF, 
                       method = "ranger",
                       #tuneLength = 6,
                       seed = 135, ## For reproducibility! ##
                       trControl = caret::trainControl(method = "repeatedcv",
                                                       number = 5,
                                                       repeats = 5,
                                                       verboseIter = T))
print(model2)

## Let's get the 'test' dataset again ##
testDF <- full_df2[8694:12970,]
summary(testDF)

##  Predicted survival (0,1) for test dataset ##
testDF$Transported <- predict(model2, testDF, type = "raw")

## Getting the final solution ##
solution <- data.frame(PassengerId = testDF$PassengerId, Transported = testDF$Transported)

## Recoding 'Transported' from '0/1' to 'True/False' ##
solution2 <- solution %>% 
             mutate(Transported = factor(case_when(Transported == 0 ~ 'False', Transported == 1 ~ 'True')))


## Export .csv file with predictions ##
readr::write_csv(x = solution2, file = "submission.csv")

## The end
```
