{"cells":[{"metadata":{"_uuid":"5d94382dffcf00cc5bd2c9b0f7a8790ba1df226e","_cell_guid":"198b0314-6e12-431e-b4e6-bdfa45660920"},"cell_type":"markdown","source":"I am learning using the Titanic data."},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nimport os\nprint(os.listdir(\"../input\"))\n\n# Any results you write to the current directory are saved as output.","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"# IMPORTING THE GOOD STUFF\nimport pandas as pd\nimport numpy as np\nimport random as rnd\n\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\ntrain_dataframe=pd.read_csv(\"../input/train.csv\")\ntest_dataframe=pd.read_csv(\"../input/test.csv\")\nprint(\"Got them!\")","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"0b9e2e1b4a5441fb6ed784424b0b07977f4b8bc5","_cell_guid":"2272d974-2271-4981-ada5-25fb6c389046","trusted":true},"cell_type":"code","source":"# Visualisation by pivoting\nprint(train_dataframe[['Pclass', 'Survived']].groupby(['Pclass'], as_index=False).mean())\n# it puts .sort_values(by='Survived',ascending=False) at the end, but that is not necessary here","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"3ee08d7c859c05a3a828d41dbb60af2c3715729c","_kg_hide-input":true,"_cell_guid":"27bedc77-1762-4e9d-a64f-2fca9028eb4a","trusted":true},"cell_type":"code","source":"# Visualising by pivoting - trying to look at sex and survived\nsex_df=train_dataframe[['Sex','Survived']]\nprint(sex_df.groupby(['Sex'], as_index=False).head())\nprint(\"_\"*40)\nprint(sex_df.groupby(['Sex'], as_index=False).mean())\nprint(\"_\"*40)\nprint(sex_df.groupby(['Sex'],as_index=False).max())\nprint(\"_\"*40)\nprint(sex_df.groupby(['Sex']).max())\nprint(\"_\"*40)\nprint(sex_df.groupby(['Sex']).std())\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c073e1bcc3a13eefb828a053364eb07de5fb793b"},"cell_type":"code","source":"agecut=pd.cut(train_dataframe['Age'], 6)\nagecut","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"0686ef1184a120fc9ea8d733e55d1711c688d704","_cell_guid":"2c6eb30a-b59b-40e5-a753-0b4201f41d01","trusted":true},"cell_type":"code","source":"survived=train_dataframe[train_dataframe['Survived']==1]\ndied=train_dataframe[train_dataframe['Survived']==0]\n# Visualisation - plots - Age\n# what is on the y axis? Frequency density?\ng = sns.FacetGrid(train_dataframe, col='Embarked')\ng.map(sns.boxplot, 'Survived', 'Age')\n\nports=train_dataframe['Embarked'].unique()\nsurv1 = survived[survived['Embarked']==ports[0]][['Age']]\nsurv2 = survived[survived['Embarked']==ports[1]]\nsurv3 = survived[survived['Embarked']==ports[2]]\ndie1 = died[died['Embarked']==ports[0]]['Age']\ndie2 = died[died['Embarked']==ports[1]]\ndie3 = died[died['Embarked']==ports[2]]\n\nh = plt.figure(figsize=(100,5))\nhax1=h.add_axes([0,0,.3, 1])\nhax1.set_xlim(0,100)\nplt.hist(surv1[['Age']], bins=40, color='green', alpha=.3)\nplt.hist(die1, bins=40, color='red', alpha=.3)\nprint(surv1.info())\nhax1.set_title(str(ports[0]))\nhax2=h.add_axes([.35, 0, .3, 1])\nhax2.set_xlim(0,100)\nplt.hist(surv2[['Age']], bins=40, color='green', alpha=.3)\nplt.hist(die2[['Age']], bins=40, color='red', alpha=.3)\nhax2.set_title(str(ports[1]))\nhax3=h.add_axes([.7,0,.3,1])\nhax3.set_xlim(0,100)\nplt.hist(surv3[['Age']], bins=40, color='green', alpha=.3)\nplt.hist(die3[['Age']], bins=40, color='red', alpha=.3)\nhax3.set_title(str(ports[2]))\n\n#then for each one, plot a histogram of died and survived by age\n#make the limits the same for all axes\n\n\n\nsh = sns.FacetGrid(train_dataframe, col='Embarked')\nsh.map(plt.hist, 'Age', bins=20)\n\ns = sns.FacetGrid(train_dataframe,col='Survived')\ns.map(plt.hist,'Age',bins=30)\n\n# what about plotting Age against chance of survival? THIS DOESN'T WORK\n# age_df=train_dataframe[['Age','Survived']].groupby(['Age']).mean()\n# t = sns.FacetGrid(age_df,col='Survived')\n# t.map(plt.hist,'Age',bins=80)","execution_count":null,"outputs":[]},{"metadata":{"collapsed":true,"_uuid":"18f81d0968c10f216c236c962f50e4d4b8d8e6b6","_cell_guid":"b667f14a-4f5a-4b9b-a519-a69f34acb93d","trusted":false},"cell_type":"code","source":"# My pivots to try and see:\n# did groups survive and die together?\n# this is not very helpful. Really I want # people on ticket/in cabin\n# but when the ticket is split across 2 or more cabins, they do seem to mostly die or mostly survive\n#is this truncating the dataset to remove those who didn't have cabins?\ncabin_df=train_dataframe[['Cabin','Age','Ticket','Survived']]\ncabin_df.groupby(['Ticket','Cabin'],as_index=False).mean()","execution_count":null,"outputs":[]},{"metadata":{"collapsed":true,"scrolled":false,"_uuid":"b8c440caa63ffb85537e6b1fe9ee065e24b7bb17","_cell_guid":"721201f7-d900-4626-8945-5c238e8a9e30","trusted":false},"cell_type":"code","source":"# Making an array of histograms to compare class and age\n# size is the size but if it's very big then the labels just get very small\n# aspect is the scale factor from the vertical to horizontal axis (regardless of units etc, just aspect\n# ratio of the histogram)\n# not filling them in provides a reasonable graph which is square\nage_Pclass_grid = sns.FacetGrid(train_dataframe, col='Survived', row='Pclass', size=2.2, aspect = 1.6)\nage_Pclass_big = sns.FacetGrid(train_dataframe, col='Survived', row='Pclass')\nage_Pclass_skinny=sns.FacetGrid(train_dataframe,col='Survived',row='Pclass', size=2.2, aspect = 9)\nA = age_Pclass_grid.map(plt.hist, 'Age', alpha=.5, bins=20)\nB = age_Pclass_grid.map(plt.hist, 'Age', alpha=.1, bins=20)\n# creating 3 in this way doesn't do anything. You only get one. I think you have to give the grids \n# different names to make different plots\nC = age_Pclass_grid.map(plt.hist, 'Age', alpha=.5, bins=20)\nage_Pclass_grid.add_legend();\n# alpha is the strength of the colour of the bars 0 - 1; default is 1 I think\n# bins is the number of classes, I think default is 10\nage_Pclass_big.map(plt.hist, 'Age', alpha=1, bins=20)\nage_Pclass_skinny.map(plt.hist, 'Age')","execution_count":null,"outputs":[]},{"metadata":{"collapsed":true,"_uuid":"58ff414d650fb19048d508193b8e66554a8c2f0e","_cell_guid":"cc188ddb-9957-4b5b-98ba-211745954c35","trusted":false},"cell_type":"code","source":"# More visualisations also part port of embark\n# grid = sns.FacetGrid(train_df, col='Embarked')\ngrid = sns.FacetGrid(train_dataframe, row='Embarked', size=2.2, aspect=1.6)\ngrid.map(sns.pointplot, 'Pclass', 'Survived', 'Sex', palette='deep')\ngrid.add_legend()\n# Now I can do what I wanted to do before with the age, since this gives probability of survival\n# on the y axis\n# So I wanted age and survived only, maybe I can split that by the 3 classes\nage_survive= sns.FacetGrid(train_dataframe,aspect=5)\nage_survive.map(sns.pointplot,'Age','Survived','Pclass', palette='deep')\nage_survive.add_legend()\n# bit of a mess because age is not scaled uniformly along the axis - the spaces between age < 1 are too big\nage_histogram = sns.FacetGrid(train_dataframe,aspect=5)\n# it doesn't work; it only takes one argument, 'Survived' it thinks it the number of bins.\n# age_histogram.map(plt.hist,'Age','Survived', bins=40)","execution_count":null,"outputs":[]},{"metadata":{"collapsed":true,"_uuid":"c241f5c783cc0656808655b8a41539a3fa05c56a","_cell_guid":"1e76fb41-8314-44e8-9640-a143265bcfb2","trusted":false},"cell_type":"code","source":"# Trying to use histogram for grouped data\n# I want to make my histogram by 'Age' with probability of survival on the y axis\n# I would like to be able to split it by 'Sex' or 'Class'\n# instead of splitting it into different histograms by whether or not they survived\n# But it doesn't work at all; what's supposed to go on the y axis? Not count/frequency any more\nage_sex_class=train_dataframe.groupby(['Pclass','Sex','Age']).mean()\nasc_grid = sns.FacetGrid(age_sex_class, aspect=3)\nasc_grid.map(plt.hist,'Age')","execution_count":null,"outputs":[]},{"metadata":{"collapsed":true,"_uuid":"48f28b13999a6b501fa6d89f32ab90e0bcad4836","_cell_guid":"6f24c7f5-3262-4d27-ac25-20f2dcd363e1","trusted":false},"cell_type":"code","source":"# experimenting with jointplot\nsns.jointplot(x='Age',y='Survived',data=train_dataframe)\nsns.jointplot(x='Age',y='Survived',kind='hex',data=train_dataframe)\nsns.jointplot(x='Age',y='Pclass',data=train_dataframe)\n# still doesn't do what I want to do - showing frequency not probabilities, so in either case\n# people in their 20s have highest survival density because there are most of them\n","execution_count":null,"outputs":[]},{"metadata":{"collapsed":true,"_uuid":"74298c3b6dac07b4afb47ce5661bdbc421864977","_cell_guid":"ff048e9b-b3de-497a-b8a3-9e60c7ac075f","trusted":false},"cell_type":"code","source":"# experimenting with barplot, pointplot for subsets of data frame\n# another attempt using the line idea\nchild_survive= sns.FacetGrid(train_dataframe[train_dataframe['Age']<18],aspect=5)\nchild_survive.map(sns.barplot,'Age','Survived','Sex', palette='deep',ci=False)\n# height of bar showns mean and line is 95% confidence interval\nchild_survive.add_legend()\n# can't restrict on both sides - think I need to redefine whole data frame\n#young_adult_survive= sns.FacetGrid(train_dataframe[35>train_dataframe['Age'] and train_dataframe['Age']>=18],aspect=5)\n#young_adult_survive.map(sns.barplot,'Age','Survived','Sex')\n#mid_adult_survive= sns.FacetGrid(train_dataframe[50>train_dataframe['Age']>=35],aspect=5)\n#mid_adult_survive.map(sns.barplot,'Age','Survived','Sex')\n#old_adult_survive= sns.FacetGrid(train_dataframe[65>train_dataframe['Age']>=50],aspect=5)\n#old_adult_survive.map(sns.barplot,'Age','Survived','Sex')\n","execution_count":null,"outputs":[]},{"metadata":{"collapsed":true,"_uuid":"d3aa45ed331999dc50b6ef68120a4e55a16d16f2","_cell_guid":"adb48cf8-b43b-40f1-9371-f642a244d854","trusted":false},"cell_type":"code","source":"# want to see the people who died on the graph but that didn't work\noldest_adult_survive=sns.barplot('Age','Survived',data=train_dataframe[train_dataframe['Age']>=60])\noldest_adult_survive.set_ylim(-1,1)","execution_count":null,"outputs":[]},{"metadata":{"collapsed":true,"_uuid":"55f5c0da572edbabf423ef67fad40a283bc80b6b","_cell_guid":"abb32c4c-c63d-4383-81d0-9a6454d0e4d8","trusted":false},"cell_type":"code","source":"# experimenting with barplot, pointplot for subsets of data frame\nsns.pointplot('Age','Survived','Sex',data=train_dataframe[train_dataframe['Age']>=60],ci=False)","execution_count":null,"outputs":[]},{"metadata":{"collapsed":true,"_uuid":"d86251335831858d6afc6c3c60a62cfb298e46b5","_cell_guid":"4f9b55f2-91bd-4b82-917e-ec46847475ec","trusted":false},"cell_type":"code","source":"# experimenting with barplot, pointplot for subsets of data frame\nsns.pointplot('Age','Survived',data=train_dataframe[train_dataframe['Age']>=60],ci=False)","execution_count":null,"outputs":[]},{"metadata":{"collapsed":true,"scrolled":false,"_uuid":"2c44ae76ac70c42ce8f0dbd5c0cbccae2015bc3e","_cell_guid":"188f80d6-cdf1-45fc-b5e6-d1fd265945ec","trusted":false},"cell_type":"code","source":"# from tutorial - categorical plots\n# I don't think this is very useful\n# I think this is better summed up in the pivots below\ngrid = sns.FacetGrid(train_dataframe, row='Embarked', col='Survived', aspect=1.6)\n# something strange is happening here; this isn't right\ngrid.map(sns.countplot, 'Sex', alpha=.5)\ngrid.add_legend()\ngrid2 = sns.FacetGrid(train_dataframe, row='Embarked', col='Survived', aspect=1.6)\ngrid2.map(sns.barplot, 'Fare', 'Sex', alpha=.5, ci=None)\ngrid2.add_legend()\ngrid3 = sns.FacetGrid(train_dataframe, row='Embarked', col='Survived', aspect=1.6)\ngrid3.map(sns.barplot,'Fare', alpha=.5, ci=None)\ngrid3.add_legend()","execution_count":null,"outputs":[]},{"metadata":{"collapsed":true,"scrolled":true,"_uuid":"2b9609dbff0d4b2c8b9d207c1883426908409432","_cell_guid":"42da733d-f8a1-47b2-85ab-60fe7f912619","trusted":false},"cell_type":"code","source":"# Checking whether I managed to show what I wanted to\ntrain_dataframe[train_dataframe['Embarked']=='S'][train_dataframe['Sex']=='female'].describe()","execution_count":null,"outputs":[]},{"metadata":{"collapsed":true,"_uuid":"660694e4c6d1dee2226899d334d643f7db8ae75c","_cell_guid":"960d6c08-c35a-45fe-b859-2912849f0419","trusted":false},"cell_type":"code","source":"# Pivoting to show different information\nprint(train_dataframe[['Survived','Embarked']].groupby(['Embarked']).mean())\nprint('_'*50)\nprint(train_dataframe[['Fare','Survived','Embarked']].groupby(['Embarked','Survived']).mean())\nprint('_'*50)\nprint(train_dataframe[['Fare','Survived','Embarked','Sex']].groupby(['Embarked','Sex','Survived']).mean())\nprint('_'*50)\n# does this do what I think it does?\nprint(train_dataframe[['Fare','Survived','Embarked','Sex']].groupby(['Embarked','Sex','Survived']).count())\n# IT DOES! SO NOW I CAN DO WHAT I WANTED TO DO FOR CABIN AND TICKET","execution_count":null,"outputs":[]},{"metadata":{"collapsed":true,"_uuid":"31cdc415afb0aa0cff9cf2fbb0572344a0168f86","_cell_guid":"bdb99374-c743-457a-8d02-e34ab0d98b18","trusted":false},"cell_type":"code","source":"# checking whether iit is showing what I wanted\ntrain_dataframe[train_dataframe['Embarked']=='S'][train_dataframe['Sex']=='male'][train_dataframe['Survived']==1].describe()","execution_count":null,"outputs":[]},{"metadata":{"collapsed":true,"_uuid":"7893c4650dd61c835d8e602f6191dee261de09d2","_cell_guid":"b6e4b92e-6afe-4c11-9527-de5300069183","trusted":false},"cell_type":"code","source":"# more visualisations pivoting by ticket, cabin\ntrain_dataframe[['Ticket','Survived','Cabin']].groupby(['Cabin','Survived']).count().sort_values(by='Ticket')","execution_count":null,"outputs":[]},{"metadata":{"collapsed":true,"scrolled":false,"_uuid":"6de322feeece3e11c921f3556d6b7ebd910470a0","_cell_guid":"f5ab226f-f3f2-45b5-ae2f-915576257559","trusted":false},"cell_type":"code","source":"# more visualisations pivoting by ticket, cabin\ntrain_dataframe[['Ticket','Survived','Cabin','Sex']].groupby(['Ticket','Survived']).count().sort_values(by='Sex')\n# really I want to sort by count(Ticket) only\n# but this doesn't work:\n# train_dataframe[['Ticket','Survived','Cabin','Sex']].groupby(['Ticket','Survived']).count().sort_values(by=['Ticket'].count('Sex'))","execution_count":null,"outputs":[]},{"metadata":{"collapsed":true,"_uuid":"b5fb718f6582c74a90655fe0bd2f30b1eaa11d8f","_cell_guid":"0ab23e51-7472-4e8a-8955-842b0bb614ec","trusted":false},"cell_type":"code","source":"# checking what the above shows\ntrain_dataframe[train_dataframe['Ticket']=='PC 17755']\n# I saw above that there were 3 'Sex' but only 2 'Cabin'; \n# This is true - Miss. Anna Ward's cabin is not listed","execution_count":null,"outputs":[]},{"metadata":{"collapsed":true,"scrolled":false,"_uuid":"ed2643e2678d507a9ae0180e2726430975725ac4","_cell_guid":"03e73aff-0087-4788-8ae4-844acce7a77b","trusted":false},"cell_type":"code","source":"# experimenting with new plot types\nsns.violinplot(x='Pclass',y='Age',hue='Survived',data=train_dataframe, split=True)\n# I was hoping this would show the relative sizes of the datasets, but actually \n# it doesn't appear to offer anything different to the plots below","execution_count":null,"outputs":[]},{"metadata":{"collapsed":true,"_uuid":"9ddcd1475158f9e838833f29ead4f79f4e3c9bda","_cell_guid":"03b57396-45ef-4afd-9f42-e133a5ef9f0a","trusted":false},"cell_type":"code","source":"# experimenting with new plot types\nsns.violinplot(x='Pclass',y='Age',hue='Survived',data=train_dataframe)","execution_count":null,"outputs":[]},{"metadata":{"collapsed":true,"_uuid":"f2d9e19ceba3aa6667171f13647d0b514f8602df","_cell_guid":"7f3e4405-ef09-47f3-8b8c-ff873302df5f","trusted":false},"cell_type":"code","source":"# experimenting with new plot types\nsns.boxplot('Survived','Age','Pclass',train_dataframe)\nsns.violinplot('Survived','Age','Pclass',train_dataframe)","execution_count":null,"outputs":[]},{"metadata":{"collapsed":true,"_uuid":"bf98c1eed9379d8141ef56a3000ac9eca1a3c5d4","_cell_guid":"a891f7cb-774f-423d-be8c-568d5ec18193","trusted":false},"cell_type":"code","source":"# experimenting with new plot types\nsns.violinplot('Survived','Age','Sex',train_dataframe)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"26486937a4c1ce3bbce4f403e6bd31494049fcff","_cell_guid":"01e690ce-3136-4f49-aa34-cfa7ae224d15","trusted":false,"collapsed":true},"cell_type":"code","source":"# how many people in the test data were on the same tickets as those in the training data?\nt=0\nfor i in train_dataframe['Ticket']:\n    for j in test_dataframe['Ticket']:\n        if i==j:\n            t+=1\nprint(t)\nc=0\nfor i in train_dataframe['Cabin']:\n    if i=='NaN':\n        break\n    for j in test_dataframe['Ticket']:\n        if i==j:\n            c+=1\nprint(c)\n# So 297 people in the test data shared tickets with people in the training data - why so many?\n# And why did noone share a cabin with anyone in the training data?","execution_count":null,"outputs":[]},{"metadata":{"collapsed":true,"_uuid":"5c5590847dd608cbb5082434ee130a557b0b3171","trusted":false},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"}},"nbformat":4,"nbformat_minor":1}