import numpy as np
import pandas as pd
import pylab as pl

#Print you can execute arbitrary python code
train = pd.read_csv("../input/train.csv", dtype={"Age": np.float64}, )
test = pd.read_csv("../input/test.csv", dtype={"Age": np.float64}, )

#Print to standard output, and see the results in the "log" section below after running your script
print("\n\nTop of the training data:")
print(train.head())

print("\n\nSummary statistics of training data")
print(train.describe())

train['Gender'] = train.Sex.map({'male':1, 'female':0}).astype(int)
#fills in missing values for age
median_ages = np.zeros((2,3))
for i in range(2):
    for j in range(3):
        median_ages[i,j] = train[(train.Gender == i) & (train.Pclass ==j+1) ]['Age'] \
                            .dropna().median()
print(median_ages)
train['Agefill'] = train['Age']
for i in range(2):
    for j in range(3):
        train.loc[(train.Age.isnull()) & (train.Pclass == j+1) &(train.Gender == i),\
        'Agefill'] = median_ages[i,j]
train['Agefill'].hist()
pl.show()

#Any files you save will be available in the output tab below
train.to_csv('copy_of_the_training_data.csv', index=False)