{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"","metadata":{"_cell_guid":"00376da6-7312-9a48-fabf-179ddc7cbdcb"}},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nfrom IPython.display import display\nimport seaborn as sns\nimport matplotlib.pylab as plt\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nfrom subprocess import check_output\nprint(check_output([\"ls\", \"../input\"]).decode(\"utf8\"))\n\n# Any results you write to the current directory are saved as output.","metadata":{"_cell_guid":"1d84de41-6154-e789-9140-3b29bbad2d15"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load training data into pandas and print first few rows and what the headers are\ntrainDF = pd.read_csv('../input/train.csv')\nprint(trainDF.columns.values)\ntrainDF.head(5)","metadata":{"_cell_guid":"d08762fe-daa9-f9c1-55fc-5cc634f3c2d5"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load test data into pandas and print first few rows and what the headers are\ntestDF = pd.read_csv('../input/test.csv')\nprint(testDF.columns.values)\ntestDF.head(5)","metadata":{"_cell_guid":"d84abaf8-9ee3-9902-71b6-4fc6cb45cf82"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Now can decide which features are categorical or numerical based on output above\n# Categorical: Sex (str), Survived (0,1), Embarked (str) - Ordinal: Pclass (1,2,3)\n# Numerical: Age (float), Parch (int), Fare (float), SibSp (int)\n# Mixed: Cabin (str), Ticket (str\n# Name contains titles like Mr., Mrs., Dr. etc.\n\n# Show distributions of numeric and objects (categorial)\ndisplay(trainDF.describe()); display( trainDF.describe(include=['O']) )","metadata":{"_cell_guid":"8e7bba73-9480-cc72-83c8-bc3e72486b51"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# In the above we see that Age and Cabin have missing/empty values\n\n# Now get some early statistics/relationships\ndisplay( trainDF['Survived'].groupby(trainDF['Pclass']).describe() ) # by class\ndisplay( trainDF['Survived'].groupby(trainDF['Sex']).describe() ) # by Sex","metadata":{"_cell_guid":"b72aa677-23a5-36a1-1098-685694c8d2fd"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Can see above that Class 1 then 2 then 3 had descending survival rates, \n#  and women had better chance of surviving then men\n# let's make this more compact\ndisplay ( trainDF[['Pclass','Survived']].groupby(['Pclass'],as_index=False).mean().sort_values(by='Survived')  )\ndisplay ( trainDF[['SibSp','Survived']].groupby(['SibSp'],as_index=False).mean().sort_values(by='Survived')  )\ndisplay ( trainDF[['Sex','Survived']].groupby(['Sex'],as_index=False).mean().sort_values(by='Survived')  )\ndisplay ( trainDF[['Parch','Survived']].groupby(['Parch'],as_index=False).mean().sort_values(by='Survived'))\ndisplay ( trainDF[['Embarked','Survived']].groupby(['Embarked'],as_index=False).mean().sort_values(by='Survived'))","metadata":{"_cell_guid":"77053c63-2203-ee46-3c1d-07c05786cf45"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Pclass definitely correlated and Sex but SibSp and Parch not really\n\n## Now look at numerical categories\n#[ trainDF[trainDF[ 'Survived'] ==S].hist(column='Age',bins=20) for S in [0,1] ]\ng=sns.FacetGrid(trainDF,col='Survived')\ng.map(plt.hist,'Age',bins=20)","metadata":{"_cell_guid":"de0f9ee7-a545-df13-3f83-0b863af026ca"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# children <5 had high survival rate\n# Combine multipple features for identifying correlations for numerical\ngrid=sns.FacetGrid(trainDF,col='Survived',row='Pclass',size=2.2,aspect=1.6)\ngrid.map(plt.hist,'Age',bins=20);grid.add_legend()","metadata":{"_cell_guid":"7919e940-bc68-b04a-04fb-bc2f56f23a9d"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Look at correlations with categorial features\ngrid=sns.FacetGrid(trainDF,row='Embarked',size=2.2,aspect=1.6)\ngrid.map(sns.pointplot,'Pclass','Survived','Sex',palette='deep')\ngrid.add_legend()","metadata":{"_cell_guid":"91e24dbd-1ec4-6a70-fd72-3d2759e57a5c"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"grid = sns.FacetGrid(trainDF, row='Embarked', col='Survived', size=2.2, aspect=1.6)\ngrid.map(sns.barplot, 'Sex', 'Fare', alpha=.5, ci=None)\ngrid.add_legend()","metadata":{"_cell_guid":"e8598b73-27be-72bf-dae3-4a11c2d4a523"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# As seen in previous part where embarked is correlated with survival more likely from C\n\n# Let's drop Ticket and Cabin from frame\ntrainDF = trainDF.drop(['Ticket','Cabin'],axis=1)\ntestDF  = testDF.drop(['Ticket','Cabin'],axis=1)\ntrainDF.head(5)","metadata":{"_cell_guid":"52090bc0-0460-fc97-7a5e-5954125c948d"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Combine datasets\ncombined=[trainDF,testDF]\n# Also extract title from Names and make it a new feature\nfor df in combined:\n    df['Title'] = df.Name.str.extract(' ([A-Za-z]+)\\.', expand=False)\n   \npd.crosstab(trainDF['Title'],trainDF['Sex'])","metadata":{"_cell_guid":"166a05a8-408d-757c-aa8f-b867792e2cc5"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# replace rare titles with common classifier rare\nfor df in combined:\n    df['Title'] = df['Title'].replace(['Lady','Countess','Capt','Col','Don','Dona','Dr','Major','Rev','Sir','Jonkheer'],'Rare')\n    df['Title'] = df['Title'].replace('Mme','Mrs')\n    df['Title'] = df['Title'].replace('Ms','Miss')\n    df['Title'] = df['Title'].replace('Mlle','Miss')\n\ntrainDF[['Title','Survived']].groupby(['Title'],as_index=False).mean()","metadata":{"_cell_guid":"3d0ce57c-ce5d-5dd9-5178-e68b4b86c1dc"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Convert Title to categorical ordinal\ntitle_dict = {\"Mr\": 1, \"Miss\": 2, \"Mrs\": 3, \"Master\": 4, \"Rare\": 5}\nfor df in combined:\n    df['Title']=df['Title'].map(title_dict)\n    df['Title']=df['Title'].fillna(0)\n\ntrainDF.head(10)","metadata":{"_cell_guid":"aa1d175a-559b-85a5-70d7-a87329505ee6"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Drop passengerID and Name\ntrainDF = trainDF.drop(['Name','PassengerId'],axis=1)\ntestDF = testDF.drop(['Name'],axis=1)\ncombined= [trainDF, testDF]\ntrainDF.head(5)","metadata":{"_cell_guid":"9a3c2218-8788-a958-0dc9-98a3d109cade"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Convert Sex to Categorical \nsex_dict={'male':0 ,'female':1}\nfor df in combined:\n    df['Sex'] = df['Sex'].map(sex_dict).astype(int)","metadata":{"_cell_guid":"61112a6f-9c8e-2caa-f006-de3a88f77eea"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## completing numerical values that are missing or NaN\n# use correlations among age, gender and Pclass along with mean and standard deviation of Age\ngrid=sns.FacetGrid(trainDF,row='Pclass',col='Sex',size=2.2,aspect=1.6)\ngrid.map(plt.hist,'Age',alpha=0.5,bins=20);grid.add_legend\n\n","metadata":{"_cell_guid":"771e9aaa-edaa-440a-89bd-d3cf7be10dbe"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Array for guessing ages\ncombined= [trainDF, testDF]\nguess_ages=np.zeros((2,3)) # sex and pclass\nfor df in combined:\n    for i in range(0,2):\n        for j in range(0,3):\n            guess_df= df[(df['Sex']==i)&(df['Pclass']==j+1)]['Age'].dropna()\n            \n            age_guess= guess_df.median()\n            guess_ages[i,j] = int(age_guess/0.5 +0.5)*0.5\n     \n            df.loc[(df.Age.isnull() ) & (df.Sex==i) & (df.Pclass==j+1),'Age' ]=guess_ages[i,j]\n            \n    df['Age'] = df['Age'].astype(int)","metadata":{"_cell_guid":"47ebd6b1-8dfa-a7a4-1d98-2c193682917c"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"trainDF.head()","metadata":{"_cell_guid":"9f2e56b4-e090-4910-7dc9-f424fdc46936"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create age bands to correlate with survived\ntrainDF['AgeBand'],agebins = pd.cut(trainDF['Age'],5,retbins=True)\ntrainDF[['AgeBand','Survived']].groupby(['AgeBand'],as_index=False).mean().sort_values(by='AgeBand')","metadata":{"_cell_guid":"9d1bd4fc-2597-005b-db21-b57361969c58"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"_cell_guid":"a248aa36-ce07-e606-699e-6a55f4dc252e"},"execution_count":null,"outputs":[]}]}