{"cells":[{"metadata":{"_uuid":"74617c5de3627121c6ed8fe725cd645eba20ca1d","_cell_guid":"fc557572-d486-cc43-c26d-47ed818ba65e"},"cell_type":"markdown","source":"## Titanic Dataset: Exploration and Model Generation ##\n\nIn this notebook, I will show how we can use data to explore the titanic disaster, to predict which passengers survived, given some data about them.\n\n"},{"metadata":{"_uuid":"14fd213f41eff07d2978f50e7915a299888fe690","_cell_guid":"0e3255fc-5060-e245-c357-b8b0fbd3b95a","trusted":true},"cell_type":"code","source":"import os\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\nfrom subprocess import check_output\nprint(check_output([\"ls\", \"../input\"]).decode(\"utf8\"))\n","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"08cb00250231b9f6f3a71b27d50af592708305c2","_cell_guid":"2bd30ac5-a133-d5b1-3d6f-3d924b351822","trusted":true},"cell_type":"code","source":"df = pd.read_csv(os.path.join('../input', 'train.csv'))\ntest = pd.read_csv(os.path.join('../input', 'test.csv'))\n\ndf.head()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"7ecfe66175a264b1b3ee46a90763c9dbb01e9f92","_cell_guid":"38f1fee3-e6ab-62b2-fec4-959d639808ce"},"cell_type":"markdown","source":"We can drop the PassangerId column because it appears that it is just another index. The ticket & name coulumns would require Natrual Language processing, so we will drop that column for now."},{"metadata":{"_uuid":"cf3a32bbe0d57d27f79348e36f710f92a51bad8b","_cell_guid":"0efbc54d-6a93-5401-1fea-51ae58c69032","trusted":true},"cell_type":"code","source":"df = df.drop(['PassengerId', 'Name', 'Ticket'], axis = 1)\n\ndf.head()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"fb0f7c88b5faef2f6c4f8d1f9d4acec7265d0e1b","_cell_guid":"329c0ad0-20b7-d245-bac7-ae11ca42fa5e"},"cell_type":"markdown","source":"## Data Visualisation ##\nWe can now visualise the data we have, explore it for insights. First we will start with a general overview of our data, transforming columns of text to numbers so we can test all correlations. Then we can explore more correlations in depth and control for otehr factors to check their validity. For example, how class will affect the survival chance? Does the port of embarkment matter?\n\n"},{"metadata":{"_uuid":"551abb5b5a5905a8a3e6a27409baf5f0fe5f5fe0","_cell_guid":"252ad944-f827-4116-a9a5-ffafa4e123ea","trusted":true},"cell_type":"code","source":"#To explore numerical corrlations, we need a number to work with.\n#We will change text to a numerical value\ndf.head()    ","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"55c482c0c6911933fc5103c8647971f3f8371657","_cell_guid":"b9697f36-b937-4b17-996a-d1ea0fa0f6c0","trusted":true},"cell_type":"code","source":"sns.barplot(x=df['Pclass'] , y=df['Survived'])\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"5de4dc60cb4a2d5bddf8f4453aa0aae3100cd610","_cell_guid":"ed4587f9-7e0d-4959-a472-8d4474fe63f7","trusted":true},"cell_type":"code","source":"sns.barplot(x=df['Sex'] , y=df['Survived'])\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"84bd1515fa069a74b0f83648ad9d360bb26e1c11","_cell_guid":"26be0776-dd8b-4ba6-975b-bb5e4b753657","trusted":true},"cell_type":"code","source":"# Allow age_df for labels to age so plot is catergoric\nage_df = df.drop(['Pclass', 'Sex', 'SibSp', 'Parch', 'Fare', 'Cabin', 'Embarked'], axis=1) \nage_df.head()\n#rename intervals, 0-5 = baby, 6-12 = child, 13-19 = teen, 20-60 = adult, 60+ = senior\n","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"85347ee0e6fab71ad6379cdb77264c1c8ff5ca90","_cell_guid":"550e0f83-8aa0-4019-aee2-c9d4b86b64cf","trusted":true},"cell_type":"code","source":"sns.barplot(x=df['SibSp'] , y=df['Survived'])\n","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"b57a6a75566c3930a6f16c27ee5dfa5071905a0d","_cell_guid":"47aa3758-e175-4884-82b9-c171fad39e08","trusted":true},"cell_type":"code","source":"sns.barplot(x=df['Parch'] , y=df['Survived'])\n","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"6c459c49ee3121d8433d734b933abde89df81602","_cell_guid":"5605ad7f-ebbc-4ff2-80d0-6d1e4a6c9054","trusted":true,"scrolled":true},"cell_type":"code","source":"sns.barplot(x=df['Fare'] , y=df['Survived'])\n","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"f95a37325380038ca9bb8266d3aecdfb54573d39","_cell_guid":"8771abc1-e208-468e-ba68-c754d0cd673b","trusted":true},"cell_type":"code","source":"df = df.replace(to_replace='male', value =1)\ndf = df.replace(to_replace='female', value =0) #Male has value 1 and female 0\n\n#Now we can explore how each value is correlated with survival\n\nfig, ax = plt.subplots()\n\n# size A4 \nfig.set_size_inches(11.7, 8.27)\nsns.heatmap(df.corr(method='spearman'))\nplt.show()\n\ndf.corr(method='spearman')\n","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"cb8dc68c3f5db70687a4410c0cc3b9ffb2cbc95c","_cell_guid":"9be4c66b-a49f-78c7-2391-3cc568e945bc","trusted":true},"cell_type":"code","source":"sns.barplot(x='Pclass', y='Survived', hue='Sex', data=df)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"33dc70d775bbfc870bb8d53d8f2d0389c44415e1","_cell_guid":"42b7e140-7055-4a56-967f-4f0bceee7d37"},"cell_type":"markdown","source":"TODO: analyse more factors\n\nWe can see that the values that impacted survival were 'Pclass', 'Sex'. We can build a basic model to predict survival based off these factors."},{"metadata":{"_uuid":"144ff48edddb92996feca1cc014c52451c51df36","_cell_guid":"a0837b5a-192c-4383-85ea-722255aa6a0c","trusted":true},"cell_type":"code","source":"# Create a dataset of only the values we want\ndf=df.drop(['Age', 'SibSp', 'Parch', 'Fare', 'Cabin', 'Embarked'], axis=1) \ndf.head()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"2764a5be6bdc86fc5dd088b4b078f0a7489cd09d","_cell_guid":"3f47e375-f745-4bd0-8ff1-1db37f88e286","trusted":true},"cell_type":"code","source":"#We can now create our basic model\nfrom sklearn import linear_model\n\nx_train = df[['Pclass', 'Sex']][1:657]\ny_train = df[['Survived']][1:657]\n\nx_test = df[['Pclass', 'Sex']][658:-1]\ny_test = df[['Survived']][658:-1]\n\nmodel = linear_model.SGDClassifier()\nmodel.fit(x_train, y_train)\n\nresults = model.predict(x_test)\nse = pd.Series(results)\n\n#combine prediction with actual answer\nresults_df = pd.DataFrame(data={'value': list(y_test['Survived']), 'prediction': se}) \n\n#create accuracy score. Accurate value if 0,0 or 1,1. Inaccurate if 0,1 or 1,1\n\nresults_df['score'] = results_df['prediction'] + results_df['value']\nscores = results_df['score'].value_counts()\n\naccuracy = ((scores[0] + scores[2])/(scores[0] + scores[1] + scores[2]))*100\nprint(\"The accuracy value is \", accuracy, \"%\")\nscores.head()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"ead54748932c6b9e03523c498e52ffaecd2c3926","collapsed":true,"_cell_guid":"c52aea22-57d7-4a36-8266-bc5e6541a747","trusted":false},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"_is_fork":false,"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"_change_revision":0},"nbformat":4,"nbformat_minor":1}