---
title: 'Dark sea'
author: 'Ciao'
date: '22 March 2017'
output:
  html_document:
    number_sections: true
    toc: true
    fig_width: 7
    fig_height: 4.5
    theme: readable
    highlight: tango
---

This is my first stab at Kaggle.

Totally inspired by [Megan Risdal](https://www.kaggle.com/mrisdal/titanic/exploring-survival-on-the-titanic).

## Load and check data


```{r, message = FALSE}
library(readr)
library(magrittr)
library(tidyverse)
library(scales)

Train <- read_csv("../input/train.csv")
Test <- read_csv("../input/test.csv")
D <- bind_rows(Train, Test)
```
Now that our packages are loaded, let's read in and take a peek at the data.

```{r}
summary(D)
```

Ok, some variable needs type transformation.

```{r}
D$Pclass %<>% as.character()
D$Fsize <- D$SibSp + D$Parch
```

## Killing all NA's

One guy lost his ticket! And *he* is class 3 and from "S".

```{r}
filter(D, is.na(Fare))

ggplot(D[D$Pclass == 3 & D$Embarked == "S" & D$Sex == "male", ],
       aes(Fare)) +
  geom_density() +
  geom_vline(aes(xintercept = median(Fare, na.rm = T)),
             color = "red", linetype='dashed', lwd=1) +
  scale_x_continuous(labels = dollar_format())

D$Fare[1044] <- median(filter(D, Pclass == 3, Embarked == 'S', Sex == 'male')$Fare, na.rm = T)
```

Two passengers don't report their port.

```{r}
filter(D, is.na(Embarked))
filter(D, Fare == 80)

D_na_Embarked <- filter(D, !is.na(Embarked))

ggplot(D_na_Embarked, aes(Embarked, Fare, fill = Pclass)) +
  geom_boxplot() +
  geom_hline(aes(yintercept = 80), 
    colour = 'red', linetype = 'dashed')

D$Embarked[c(62, 830)] <- 'C'
```

418 passengers' age is unknown. And their age is related to other variables.

```{r}
D$AgeKnown <- ifelse(!is.na(D$Age), "known", "notknown")

prop.table(table(D$Embarked, D$AgeKnown), 2)

prop.table(table(D$Survived, D$AgeKnown), 2)

ggplot(D, aes(AgeKnown, Fare, fill = Pclass)) +
  geom_boxplot()

library(mice)
set.seed(129)

mice_mod <- mice(D[, !names(D) %in% c('PassengerId','Name','Ticket','Cabin','Family','Surname','Survived')], method = "rf") 
mice_output <- complete(mice_mod)

par(mfrow=c(1,2))
hist(D$Age, freq=F, main='Age: Original Data', 
  col='darkgreen', ylim=c(0,0.04))
hist(mice_output$Age, freq=F, main='Age: MICE Output', 
  col='lightgreen', ylim=c(0,0.04))

D$Age <- mice_output$Age

```
### Predict

```{r}
library(aod)
Test <- D[892:1309, ]
Train <- D[1:891, ]

mylogit <- glm(Survived ~ Pclass + Sex + log(Fare+1) + Fsize + Embarked + Age,
               family = binomial("logit"), data = Train)
summary(mylogit)
Survived <- predict(mylogit, Test[, c(3, 5, 6, 10, 13, 12)])
Survived <- ifelse(Survived >= 0, 1, 0)
Res <- data.frame(Test[,1], Survived)
write.csv(Res, 'titanic_pred2.csv', row.names = F)
```