{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n%matplotlib inline","metadata":{"execution":{"iopub.status.busy":"2022-07-14T09:51:01.089324Z","iopub.execute_input":"2022-07-14T09:51:01.089767Z","iopub.status.idle":"2022-07-14T09:51:01.097506Z","shell.execute_reply.started":"2022-07-14T09:51:01.089731Z","shell.execute_reply":"2022-07-14T09:51:01.096305Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data = pd.read_csv('../input/titanic/train.csv')\ntest_data = pd.read_csv('../input/titanic/test.csv')","metadata":{"execution":{"iopub.status.busy":"2022-07-14T09:51:01.116571Z","iopub.execute_input":"2022-07-14T09:51:01.117037Z","iopub.status.idle":"2022-07-14T09:51:01.134135Z","shell.execute_reply.started":"2022-07-14T09:51:01.117000Z","shell.execute_reply":"2022-07-14T09:51:01.133243Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"display(train_data.head())\ndisplay(test_data.head())","metadata":{"execution":{"iopub.status.busy":"2022-07-14T09:51:01.136433Z","iopub.execute_input":"2022-07-14T09:51:01.137667Z","iopub.status.idle":"2022-07-14T09:51:01.169622Z","shell.execute_reply.started":"2022-07-14T09:51:01.137618Z","shell.execute_reply":"2022-07-14T09:51:01.168246Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"display(train_data.info())\ndisplay(test_data.info())","metadata":{"execution":{"iopub.status.busy":"2022-07-14T09:51:01.171327Z","iopub.execute_input":"2022-07-14T09:51:01.172469Z","iopub.status.idle":"2022-07-14T09:51:01.199502Z","shell.execute_reply.started":"2022-07-14T09:51:01.172433Z","shell.execute_reply":"2022-07-14T09:51:01.198625Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data = train_data.drop('Cabin',axis=1)\ntest_data = test_data.drop('Cabin',axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-07-14T09:51:01.201444Z","iopub.execute_input":"2022-07-14T09:51:01.201941Z","iopub.status.idle":"2022-07-14T09:51:01.210416Z","shell.execute_reply.started":"2022-07-14T09:51:01.201911Z","shell.execute_reply":"2022-07-14T09:51:01.209410Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data['Embarked'].fillna(method = 'ffill',inplace=True)\ntest_data['Fare'].fillna(method = 'ffill',inplace=True)\ndisplay(train_data.info())\ndisplay(test_data.info())","metadata":{"execution":{"iopub.status.busy":"2022-07-14T09:51:01.211864Z","iopub.execute_input":"2022-07-14T09:51:01.212233Z","iopub.status.idle":"2022-07-14T09:51:01.244837Z","shell.execute_reply.started":"2022-07-14T09:51:01.212198Z","shell.execute_reply":"2022-07-14T09:51:01.243757Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data = train_data.drop('Name',axis=1)\ntest_data = test_data.drop('Name',axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-07-14T09:51:01.246192Z","iopub.execute_input":"2022-07-14T09:51:01.246624Z","iopub.status.idle":"2022-07-14T09:51:01.254415Z","shell.execute_reply.started":"2022-07-14T09:51:01.246588Z","shell.execute_reply":"2022-07-14T09:51:01.252918Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data = pd.get_dummies(train_data,columns=['Sex'],drop_first = True)\ntrain_data.rename(columns = {'Sex_male':'Sex'}, inplace = True)\ntest_data = pd.get_dummies(test_data,columns=['Sex'],drop_first = True)\ntest_data.rename(columns = {'Sex_male':'Sex'}, inplace = True)","metadata":{"execution":{"iopub.status.busy":"2022-07-14T09:51:01.255917Z","iopub.execute_input":"2022-07-14T09:51:01.257010Z","iopub.status.idle":"2022-07-14T09:51:01.275448Z","shell.execute_reply.started":"2022-07-14T09:51:01.256973Z","shell.execute_reply":"2022-07-14T09:51:01.274472Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"display(train_data['Ticket'].nunique())\ndisplay(test_data['Ticket'].nunique())","metadata":{"execution":{"iopub.status.busy":"2022-07-14T09:51:01.277060Z","iopub.execute_input":"2022-07-14T09:51:01.277419Z","iopub.status.idle":"2022-07-14T09:51:01.287480Z","shell.execute_reply.started":"2022-07-14T09:51:01.277389Z","shell.execute_reply":"2022-07-14T09:51:01.286233Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data = train_data.drop('Ticket',axis=1)\ntest_data = test_data.drop('Ticket',axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-07-14T09:51:01.290077Z","iopub.execute_input":"2022-07-14T09:51:01.291005Z","iopub.status.idle":"2022-07-14T09:51:01.299183Z","shell.execute_reply.started":"2022-07-14T09:51:01.290966Z","shell.execute_reply":"2022-07-14T09:51:01.298267Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.loc[train_data['Embarked'] == 'S', 'Embarked'] = 0\ntrain_data.loc[train_data['Embarked'] == 'C', 'Embarked'] = 1\ntrain_data.loc[train_data['Embarked'] == 'Q', 'Embarked'] = 2\ntest_data.loc[test_data['Embarked'] == 'S', 'Embarked'] = 0\ntest_data.loc[test_data['Embarked'] == 'C', 'Embarked'] = 1\ntest_data.loc[test_data['Embarked'] == 'Q', 'Embarked'] = 2","metadata":{"execution":{"iopub.status.busy":"2022-07-14T09:51:01.342095Z","iopub.execute_input":"2022-07-14T09:51:01.342797Z","iopub.status.idle":"2022-07-14T09:51:01.356424Z","shell.execute_reply.started":"2022-07-14T09:51:01.342746Z","shell.execute_reply":"2022-07-14T09:51:01.355402Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.to_numeric(train_data['Embarked'])\ntrain_data['Embarked'] = train_data['Embarked'].astype(int)\npd.to_numeric(test_data['Embarked'])\ntest_data['Embarked'] = test_data['Embarked'].astype(int)","metadata":{"execution":{"iopub.status.busy":"2022-07-14T09:51:01.358664Z","iopub.execute_input":"2022-07-14T09:51:01.359650Z","iopub.status.idle":"2022-07-14T09:51:01.372261Z","shell.execute_reply.started":"2022-07-14T09:51:01.359584Z","shell.execute_reply":"2022-07-14T09:51:01.370782Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"display(train_data.info())\ndisplay(test_data.info())","metadata":{"execution":{"iopub.status.busy":"2022-07-14T09:51:01.374059Z","iopub.execute_input":"2022-07-14T09:51:01.374751Z","iopub.status.idle":"2022-07-14T09:51:01.405800Z","shell.execute_reply.started":"2022-07-14T09:51:01.374717Z","shell.execute_reply":"2022-07-14T09:51:01.404668Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.to_numeric(train_data['Embarked'])\ntrain_data['Embarked'] = train_data['Embarked'].astype(int)\npd.to_numeric(test_data['Embarked'])\ntest_data['Embarked'] = test_data['Embarked'].astype(int)","metadata":{"execution":{"iopub.status.busy":"2022-07-14T09:51:01.407011Z","iopub.execute_input":"2022-07-14T09:51:01.407363Z","iopub.status.idle":"2022-07-14T09:51:01.415960Z","shell.execute_reply.started":"2022-07-14T09:51:01.407331Z","shell.execute_reply":"2022-07-14T09:51:01.414470Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"display(train_data.info())\ndisplay(test_data.info())","metadata":{"execution":{"iopub.status.busy":"2022-07-14T09:51:01.418410Z","iopub.execute_input":"2022-07-14T09:51:01.418730Z","iopub.status.idle":"2022-07-14T09:51:01.445993Z","shell.execute_reply.started":"2022-07-14T09:51:01.418702Z","shell.execute_reply":"2022-07-14T09:51:01.445057Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.corr(method ='pearson')['Age']","metadata":{"execution":{"iopub.status.busy":"2022-07-14T09:51:01.447236Z","iopub.execute_input":"2022-07-14T09:51:01.448112Z","iopub.status.idle":"2022-07-14T09:51:01.457737Z","shell.execute_reply.started":"2022-07-14T09:51:01.448076Z","shell.execute_reply":"2022-07-14T09:51:01.456650Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"display(train_data['SibSp'].nunique())\ndisplay(train_data['Parch'].nunique())\ndisplay(train_data['Fare'].nunique())\n","metadata":{"execution":{"iopub.status.busy":"2022-07-14T09:51:01.460281Z","iopub.execute_input":"2022-07-14T09:51:01.462438Z","iopub.status.idle":"2022-07-14T09:51:01.472559Z","shell.execute_reply.started":"2022-07-14T09:51:01.462399Z","shell.execute_reply":"2022-07-14T09:51:01.471483Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.groupby(['Pclass','Sex'], as_index=False)['Age'].mean().round(0)","metadata":{"execution":{"iopub.status.busy":"2022-07-14T09:51:01.474106Z","iopub.execute_input":"2022-07-14T09:51:01.474998Z","iopub.status.idle":"2022-07-14T09:51:01.492708Z","shell.execute_reply.started":"2022-07-14T09:51:01.474967Z","shell.execute_reply":"2022-07-14T09:51:01.491405Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def impute_age(cols):\n    Age = cols[0]\n    Pclass = cols[1]\n    Sex = cols[2]\n    \n    if pd.isnull(Age):\n\n        if Sex == 0:\n            if Pclass == 1:\n                return 35\n\n            elif Pclass == 2:\n                return 29\n            else:\n                return 23\n        else:\n            if Pclass == 1:\n                return 41\n\n            elif Pclass == 2:\n                return 31\n\n            else:\n                return 26\n\n    else:\n        return Age","metadata":{"execution":{"iopub.status.busy":"2022-07-14T09:51:01.494651Z","iopub.execute_input":"2022-07-14T09:51:01.495325Z","iopub.status.idle":"2022-07-14T09:51:01.503085Z","shell.execute_reply.started":"2022-07-14T09:51:01.495278Z","shell.execute_reply":"2022-07-14T09:51:01.502242Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data['Age'] = train_data[['Age','Pclass','Sex']].apply(impute_age,axis=1)\ntest_data['Age'] = test_data[['Age','Pclass','Sex']].apply(impute_age,axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-07-14T09:51:01.504356Z","iopub.execute_input":"2022-07-14T09:51:01.505203Z","iopub.status.idle":"2022-07-14T09:51:01.550849Z","shell.execute_reply.started":"2022-07-14T09:51:01.505141Z","shell.execute_reply":"2022-07-14T09:51:01.549648Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"display(train_data.info())\ndisplay(test_data.info())","metadata":{"execution":{"iopub.status.busy":"2022-07-14T09:51:01.554463Z","iopub.execute_input":"2022-07-14T09:51:01.554816Z","iopub.status.idle":"2022-07-14T09:51:01.582378Z","shell.execute_reply.started":"2022-07-14T09:51:01.554782Z","shell.execute_reply":"2022-07-14T09:51:01.581308Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data = train_data.drop('PassengerId',axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-07-14T09:51:01.583881Z","iopub.execute_input":"2022-07-14T09:51:01.584293Z","iopub.status.idle":"2022-07-14T09:51:01.590979Z","shell.execute_reply.started":"2022-07-14T09:51:01.584264Z","shell.execute_reply":"2022-07-14T09:51:01.589699Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(12,6))\nheat = sns.heatmap(data=train_data.corr(), annot=True, cmap='coolwarm',alpha=0.7)\nheat.tick_params(axis='x', labelsize=14)\nheat.tick_params(axis='y', labelsize=14)\nheat.set_title('Training Set Correlations', size=10)","metadata":{"execution":{"iopub.status.busy":"2022-07-14T09:51:01.594332Z","iopub.execute_input":"2022-07-14T09:51:01.594642Z","iopub.status.idle":"2022-07-14T09:51:02.186692Z","shell.execute_reply.started":"2022-07-14T09:51:01.594614Z","shell.execute_reply":"2022-07-14T09:51:02.185610Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"survived = train_data['Survived'].value_counts()[1]\nnot_survived = train_data['Survived'].value_counts()[0]\nsurvived_per = survived / train_data.shape[0] * 100\nnot_survived_per = not_survived / train_data.shape[0] * 100\n\nplt.figure(figsize=(8, 6))\nsns.set_style('darkgrid')\nsns.countplot(x = train_data['Survived'], palette = 'coolwarm')\n\nplt.xlabel('Survival', size=12, labelpad=15)\nplt.ylabel('Passenger Count', size=15, labelpad=15)\nplt.xticks((0, 1), ['Not Survived ({0:.2f}%)'.format(not_survived_per), 'Survived ({0:.2f}%)'.format(survived_per)])\nplt.tick_params(axis='x', labelsize=13)\nplt.tick_params(axis='y', labelsize=13)\nplt.title('Training Set Survival Distribution', size=12, y=1.05)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-14T09:51:02.188150Z","iopub.execute_input":"2022-07-14T09:51:02.188554Z","iopub.status.idle":"2022-07-14T09:51:02.361871Z","shell.execute_reply.started":"2022-07-14T09:51:02.188524Z","shell.execute_reply":"2022-07-14T09:51:02.360403Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"f,ax=plt.subplots(1,2,figsize=(18,8))\nsns.set_style('darkgrid')\n\ncc = sns.countplot(x='Pclass',data = train_data, palette='coolwarm', ax = ax[0])\nfor p in cc.patches:\n        cc.annotate('{:.1f}'.format(p.get_height()), (p.get_x()+0.1, p.get_height()+5))\nax[0].set_title('Number of passengers for each class')\nax[0].set_ylabel('Passenger count')\n\ncs = sns.countplot(x='Pclass',data = train_data,hue = 'Survived', palette='coolwarm',ax=ax[1])\nfor p in cs.patches:\n        cs.annotate('{:.1f}'.format(p.get_height()), (p.get_x()+0.1, p.get_height()+5))\nax[1].set_title('Survival distribution of passengers for each class')\nax[1].set_ylabel('Passenger count')\nax[1].legend(['Not Survived', 'Survived'], loc='upper left', prop={'size': 12})\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-14T09:51:02.363197Z","iopub.execute_input":"2022-07-14T09:51:02.363658Z","iopub.status.idle":"2022-07-14T09:51:02.762148Z","shell.execute_reply.started":"2022-07-14T09:51:02.363616Z","shell.execute_reply":"2022-07-14T09:51:02.761348Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Let's explore age distribution of passengers.","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(20,16))\nsns.set_style('darkgrid')\ntotal = len(train_data['Age'])*1.\nca = sns.histplot(x='Age',hue = train_data['Survived'],data=train_data, palette = 'coolwarm')\nfor p in ca.patches:\n        ca.annotate('{:.1f}%'.format(100*p.get_height()/total), (p.get_x()+0.5, p.get_height()+2))\nca.set_ylabel('Passenger count')\nca.legend(['Survived', 'Not survived'], loc='upper right', prop={'size': 15})","metadata":{"execution":{"iopub.status.busy":"2022-07-14T09:51:02.763951Z","iopub.execute_input":"2022-07-14T09:51:02.764292Z","iopub.status.idle":"2022-07-14T09:51:03.525618Z","shell.execute_reply.started":"2022-07-14T09:51:02.764262Z","shell.execute_reply":"2022-07-14T09:51:03.524482Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Most of the passengers were in the age between 20 to 30 years. This age category also has the smallest survival rate.\n\nNow we will explore SibSp and Parch columns. Let's combine them in two subplots.","metadata":{}},{"cell_type":"code","source":"f,ax=plt.subplots(1,2,figsize=(18,8))\nsns.set_style('darkgrid')\ntotal = len(train_data['SibSp'])*1.\nca = sns.countplot(x='SibSp',hue = train_data['Survived'],data=train_data, palette='coolwarm', ax=ax[0])\nfor p in ca.patches:\n        ca.annotate('{:.1f}%'.format(100*p.get_height()/total), (p.get_x()+0.1, p.get_height()+2))\nax[0].set_title('Survival distribution of passengers for SibSp')\nax[0].set_ylabel('Passenger count')\nax[0].legend(['Not Survived', 'Survived'], loc='upper right', prop={'size': 12})\n\nca = sns.countplot(x='Parch',hue = train_data['Survived'],data=train_data, palette='coolwarm', ax=ax[1])\nfor p in ca.patches:\n        ca.annotate('{:.1f}%'.format(100*p.get_height()/total), (p.get_x()+0, p.get_height()+2))\nax[1].set_title('Survival distribution of passengers for Parch')\nax[1].set_ylabel('Passenger count')\nax[1].legend(['Not Survived', 'Survived'], loc='upper right', prop={'size': 12})","metadata":{"execution":{"iopub.status.busy":"2022-07-14T09:51:03.527221Z","iopub.execute_input":"2022-07-14T09:51:03.527880Z","iopub.status.idle":"2022-07-14T09:51:04.274298Z","shell.execute_reply.started":"2022-07-14T09:51:03.527835Z","shell.execute_reply":"2022-07-14T09:51:04.273211Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Most of the passengers were alone. Interesting, that survival rate of the passengers with 1 sibling/spouse and 1 parent/children is the highest.\n\nNext column is the Fare. Let's see passenger distribution along with the survival rate.","metadata":{}},{"cell_type":"code","source":"fig, axs = plt.subplots(figsize=(22, 9))\nsns.countplot(x=pd.qcut(train_data['Fare'], 13), hue='Survived', data=train_data, palette='coolwarm')\n\nplt.xlabel('Fare', size=15, labelpad=20)\nplt.ylabel('Passenger Count', size=15, labelpad=20)\nplt.tick_params(axis='x', labelsize=10)\nplt.tick_params(axis='y', labelsize=15)\n\nplt.legend(['Not Survived', 'Survived'], loc='upper right', prop={'size': 15})\nplt.title('Count of Survival in {} Feature'.format('Fare'), size=15, y=1.05)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-14T09:51:04.276067Z","iopub.execute_input":"2022-07-14T09:51:04.276448Z","iopub.status.idle":"2022-07-14T09:51:04.626573Z","shell.execute_reply.started":"2022-07-14T09:51:04.276416Z","shell.execute_reply":"2022-07-14T09:51:04.625538Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The higher the price of the ticket - the higher survival rate. The heatmap we created earlier showed high correlation between the Fare and the Pclass. Let's explore it then.\n\n","metadata":{}},{"cell_type":"code","source":"fig, axs = plt.subplots(figsize=(22, 9))\ncs = sns.countplot(x=pd.qcut(train_data['Fare'], 13), hue='Pclass', data=train_data, palette='coolwarm')\n\nplt.xlabel('Fare', size=15, labelpad=20)\nplt.ylabel('Passenger Count', size=15, labelpad=20)\nplt.tick_params(axis='x', labelsize=10)\nplt.tick_params(axis='y', labelsize=15)\nfor p in cs.patches:\n        cs.annotate('{:.1f}'.format(p.get_height()), (p.get_x()+0, p.get_height()+2))\n\nplt.legend(['1', '2', '3'], loc='upper right', prop={'size': 15})\nplt.title('Fare distribution between classes', size=15, y=1.05)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-14T09:51:04.628312Z","iopub.execute_input":"2022-07-14T09:51:04.628674Z","iopub.status.idle":"2022-07-14T09:51:05.186762Z","shell.execute_reply.started":"2022-07-14T09:51:04.628646Z","shell.execute_reply":"2022-07-14T09:51:05.185742Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Except 12 odd values for the 1st and the 2nd classes (within range 0-7.229), the plot confirms correlation between the cost of the ticket and the class.\n\nNow, we will explore the Embarked column.","metadata":{}},{"cell_type":"code","source":"f,ax=plt.subplots(1,2,figsize=(18,8))\nsns.set_style('darkgrid')\n\ncc = sns.countplot(x='Embarked',data = train_data, palette='coolwarm', ax = ax[0])\nfor p in cc.patches:\n        cc.annotate('{:.1f}'.format(p.get_height()), (p.get_x()+0.1, p.get_height()+5))\nax[0].set_title('Number of passengers for each port')\nax[0].set_xticklabels(['S', 'C', 'Q'])\nax[0].set_ylabel('Passenger count')\n\ncs = sns.countplot(x='Embarked',data = train_data,hue = 'Survived', palette='coolwarm',ax=ax[1])\nfor p in cs.patches:\n        cs.annotate('{:.1f}'.format(p.get_height()), (p.get_x()+0.1, p.get_height()+5))\nax[1].set_title('Survival distribution of passengers for each port')\nax[1].set_xticklabels(['S', 'C', 'Q'])\nax[1].set_ylabel('Passenger count')\nax[1].legend(['Not Survived', 'Survived'], loc='upper right', prop={'size': 12})\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-14T09:51:05.188700Z","iopub.execute_input":"2022-07-14T09:51:05.189017Z","iopub.status.idle":"2022-07-14T09:51:05.579412Z","shell.execute_reply.started":"2022-07-14T09:51:05.188988Z","shell.execute_reply":"2022-07-14T09:51:05.578242Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Most of the passengers were from the \"S\" port. Suprisingly, the \"C\" port shows higher ratio of survived. As we know for now, highest correlation with the \"Survival\" have \"Sex\" and \"Pclass\" parameters. Let's explore them.","metadata":{}},{"cell_type":"code","source":"f,ax=plt.subplots(1,2,figsize=(18,8))\nsns.set_style('darkgrid')\ncs = sns.countplot(x='Embarked',data = train_data,hue = 'Pclass', palette='coolwarm', ax=ax[0])\nfor p in cs.patches:\n        cs.annotate('{:.1f}'.format(p.get_height()), (p.get_x()+0.1, p.get_height()+5))\nax[0].set_title('Class distribution  for each port')\nax[0].set_xticklabels(['S', 'C', 'Q'])\nax[0].set_ylabel('Passenger count')\nax[0].legend(['1', '2', '3'], loc='upper right', prop={'size': 12})\n\ncs = sns.countplot(x='Embarked',data = train_data,hue = 'Sex', palette='coolwarm', ax=ax[1])\nfor p in cs.patches:\n        cs.annotate('{:.1f}'.format(p.get_height()), (p.get_x()+0.1, p.get_height()+5))\nax[1].set_title('Sex distribution  for each port')\nax[1].set_xticklabels(['S', 'C', 'Q'])\nax[1].set_ylabel('Passenger count')\nax[1].legend(['Female', 'Male'], loc='upper right', prop={'size': 12})\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-14T09:51:05.581104Z","iopub.execute_input":"2022-07-14T09:51:05.581480Z","iopub.status.idle":"2022-07-14T09:51:06.022364Z","shell.execute_reply.started":"2022-07-14T09:51:05.581450Z","shell.execute_reply":"2022-07-14T09:51:06.020983Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We can see, that the class distribution for the \"C\" port is different then for other two. Sex distribution shows, that both ports \"C\" and \"Q\" have almost equal numbers of male and female passengers while number of male passengers from \"S\" port is more than twice the number of female passengers.\n\nNow we continue with \"Sex\" column.","metadata":{}},{"cell_type":"code","source":"fig, axes = plt.subplots(1, 3, figsize=(20,8))\n\naxes[0].pie(train_data.loc[train_data['Sex'],'Sex'].value_counts(),labels = ['Male','Female'],autopct='%1.0f%%', colors = ['#ff9950','#66b3ff'],wedgeprops={'alpha':0.5})\naxes[0].set_title('Sex distribution', fontsize = 15)\n\naxes[1].pie(train_data.loc[train_data[\"Sex\"] == 1, \"Survived\"].value_counts(),labels = ['No survived','Survived'],autopct='%1.0f%%',colors = ['#ff9950','#66b3ff'],wedgeprops={'alpha':0.5})\naxes[1].set_title('% of survived male', fontsize = 15)\n\naxes[2].pie(train_data.loc[train_data[\"Sex\"] == 0, \"Survived\"].value_counts(),labels = ['Survived','No survived'], autopct='%1.0f%%',colors = ['#ff9950','#66b3ff'],wedgeprops={'alpha':0.5})\naxes[2].set_title('% of survived female', fontsize = 15)\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-14T09:51:06.023796Z","iopub.execute_input":"2022-07-14T09:51:06.024733Z","iopub.status.idle":"2022-07-14T09:51:06.316946Z","shell.execute_reply.started":"2022-07-14T09:51:06.024682Z","shell.execute_reply":"2022-07-14T09:51:06.315568Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"As it was non-directly confirmed earlier, the number of male passengers is bigger than the number of female passengers. However, the survival rate is higher for female passengers.\n\nFor the end, let's create some violin plots. For example, Pclass-Age and Sex-Age distribution with Survived as parameter.","metadata":{}},{"cell_type":"code","source":"f,ax=plt.subplots(1,2,figsize=(18,8))\nsns.set_style('darkgrid')\nsns.violinplot(x = \"Pclass\",y = \"Age\", hue=\"Survived\", data=train_data,palette='coolwarm',split=True,ax=ax[0])\nax[0].set_title('Pclass and Age vs Survived')\nax[0].set_yticks(range(0,110,10))\nax[0].legend(handles=ax[0].legend_.legendHandles, labels=['Not survived', 'Survived'])\nsns.violinplot(x = \"Sex\",y = \"Age\", hue=\"Survived\", data=train_data,palette = 'coolwarm',split=True,ax=ax[1])\nax[1].set_title('Sex and Age vs Survived')\nax[1].set_yticks(range(0,110,10))\nax[1].legend(handles=ax[1].legend_.legendHandles, labels=['Not survived', 'Survived'])\nax[1].set_xticklabels(['Female', 'Male'])\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-14T09:51:06.319048Z","iopub.execute_input":"2022-07-14T09:51:06.319842Z","iopub.status.idle":"2022-07-14T09:51:06.825999Z","shell.execute_reply.started":"2022-07-14T09:51:06.319808Z","shell.execute_reply":"2022-07-14T09:51:06.824857Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"4. Predictons\n4.1. Models import\nWe will import some base models for making predictions.","metadata":{}},{"cell_type":"code","source":"from sklearn.linear_model import LogisticRegression \nfrom sklearn.ensemble import RandomForestClassifier \nfrom sklearn.neighbors import KNeighborsClassifier \nfrom sklearn.tree import DecisionTreeClassifier \nfrom sklearn.model_selection import train_test_split \nfrom sklearn import metrics \nfrom sklearn.metrics import classification_report\nfrom sklearn.metrics import confusion_matrix ","metadata":{"execution":{"iopub.status.busy":"2022-07-14T09:51:06.827625Z","iopub.execute_input":"2022-07-14T09:51:06.827955Z","iopub.status.idle":"2022-07-14T09:51:06.833599Z","shell.execute_reply.started":"2022-07-14T09:51:06.827926Z","shell.execute_reply":"2022-07-14T09:51:06.832199Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Now we will split our train data so we can observe which model has better score.","metadata":{}},{"cell_type":"code","source":"X = train_data.drop('Survived', axis =1)\ny = train_data['Survived']\nX_train, X_test,y_train, y_test = train_test_split(X,y,test_size = 0.3, random_state = 101)","metadata":{"execution":{"iopub.status.busy":"2022-07-14T09:51:06.838507Z","iopub.execute_input":"2022-07-14T09:51:06.838863Z","iopub.status.idle":"2022-07-14T09:51:06.850956Z","shell.execute_reply.started":"2022-07-14T09:51:06.838822Z","shell.execute_reply":"2022-07-14T09:51:06.850026Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"First model will be logistic regression.","metadata":{}},{"cell_type":"code","source":"logreg = LogisticRegression()\nlogreg.fit(X_train, y_train)\ny_pred = logreg.predict(X_test)\nprint(classification_report(y_test,y_pred))\nprint(confusion_matrix(y_test,y_pred))","metadata":{"execution":{"iopub.status.busy":"2022-07-14T09:51:06.852700Z","iopub.execute_input":"2022-07-14T09:51:06.853983Z","iopub.status.idle":"2022-07-14T09:51:06.900761Z","shell.execute_reply.started":"2022-07-14T09:51:06.853934Z","shell.execute_reply":"2022-07-14T09:51:06.899539Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Logistic regression shows pretty good score. We will continue with K neighbor classifier. Additionally, we will observe for which number of neighbors the model shows the best results.\n\nStart the model with for loop.","metadata":{}},{"cell_type":"code","source":"error = []\nfor i in range (1,40):\n    \n    knn = KNeighborsClassifier(n_neighbors=i)\n    knn.fit(X_train, y_train)\n    y_predi = knn.predict(X_test)\n    error.append(np.mean(y_predi != y_test))","metadata":{"execution":{"iopub.status.busy":"2022-07-14T09:51:06.902462Z","iopub.execute_input":"2022-07-14T09:51:06.902913Z","iopub.status.idle":"2022-07-14T09:51:07.495784Z","shell.execute_reply.started":"2022-07-14T09:51:06.902867Z","shell.execute_reply":"2022-07-14T09:51:07.494612Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Now, let's print error distribution for the number of neighbors","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(10,6))\ner = plt.plot(range(1,40),error,color='blue',marker='o',markerfacecolor='orange',markersize=10)\nplt.title('Error distribution  for the different number of neighbors')\nplt.xlabel('Number of neighbors')\nplt.ylabel('Error')","metadata":{"execution":{"iopub.status.busy":"2022-07-14T09:51:07.497271Z","iopub.execute_input":"2022-07-14T09:51:07.498148Z","iopub.status.idle":"2022-07-14T09:51:07.729546Z","shell.execute_reply.started":"2022-07-14T09:51:07.498110Z","shell.execute_reply":"2022-07-14T09:51:07.728574Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The smallest error is for 7 neighbors. We will use this number for our model.","metadata":{}},{"cell_type":"code","source":"knn = KNeighborsClassifier(n_neighbors=7)\nknn.fit(X_train, y_train)\ny_pred = knn.predict(X_test)\n\nprint(classification_report(y_test,y_pred))\nprint(confusion_matrix(y_test,y_pred))","metadata":{"execution":{"iopub.status.busy":"2022-07-14T09:51:07.730850Z","iopub.execute_input":"2022-07-14T09:51:07.731172Z","iopub.status.idle":"2022-07-14T09:51:07.763246Z","shell.execute_reply.started":"2022-07-14T09:51:07.731132Z","shell.execute_reply":"2022-07-14T09:51:07.762028Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Even with the best model parameters K neighbor classifier model is worse in predictions then logistic regression model.\n\nNow, let's use the decision tree classifier model.","metadata":{}},{"cell_type":"code","source":"dtree = DecisionTreeClassifier()\ndtree.fit(X_train, y_train)\ny_pred = dtree.predict(X_test)\nprint(classification_report(y_test,y_pred))\nprint(confusion_matrix(y_test,y_pred))","metadata":{"execution":{"iopub.status.busy":"2022-07-14T09:51:07.764941Z","iopub.execute_input":"2022-07-14T09:51:07.766143Z","iopub.status.idle":"2022-07-14T09:51:07.785400Z","shell.execute_reply.started":"2022-07-14T09:51:07.766096Z","shell.execute_reply":"2022-07-14T09:51:07.784243Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"This model shows the same accuracy, precision and f1-score as logistic regression.\n\nNow, we will use our last model: the random forestt classifier.","metadata":{}},{"cell_type":"code","source":"random_forest = RandomForestClassifier(n_estimators=200)\n\nrandom_forest.fit(X_train, y_train)\n\ny_pred = random_forest.predict(X_test)\n\nprint(classification_report(y_test,y_pred))\nprint(confusion_matrix(y_test,y_pred))","metadata":{"execution":{"iopub.status.busy":"2022-07-14T09:51:07.786849Z","iopub.execute_input":"2022-07-14T09:51:07.787330Z","iopub.status.idle":"2022-07-14T09:51:08.259129Z","shell.execute_reply.started":"2022-07-14T09:51:07.787287Z","shell.execute_reply":"2022-07-14T09:51:08.258246Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The Random Forest Classifier shows the best scores. We will use it for our predictions.","metadata":{}},{"cell_type":"code","source":"X_data = test_data.drop('PassengerId', axis=1).copy()\nX_train = train_data.drop('Survived', axis =1)\ny_train = train_data['Survived']","metadata":{"execution":{"iopub.status.busy":"2022-07-14T09:51:08.260382Z","iopub.execute_input":"2022-07-14T09:51:08.261449Z","iopub.status.idle":"2022-07-14T09:51:08.268476Z","shell.execute_reply.started":"2022-07-14T09:51:08.261414Z","shell.execute_reply":"2022-07-14T09:51:08.267219Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"random_forest = RandomForestClassifier(n_estimators=200)\nrandom_forest.fit(X_train, y_train)\nY_pred = random_forest.predict(X_data)","metadata":{"execution":{"iopub.status.busy":"2022-07-14T09:51:08.269809Z","iopub.execute_input":"2022-07-14T09:51:08.270219Z","iopub.status.idle":"2022-07-14T09:51:08.771861Z","shell.execute_reply.started":"2022-07-14T09:51:08.270179Z","shell.execute_reply":"2022-07-14T09:51:08.770746Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"PassengerId = test_data['PassengerId']","metadata":{"execution":{"iopub.status.busy":"2022-07-14T09:51:08.772905Z","iopub.execute_input":"2022-07-14T09:51:08.773206Z","iopub.status.idle":"2022-07-14T09:51:08.778889Z","shell.execute_reply.started":"2022-07-14T09:51:08.773179Z","shell.execute_reply":"2022-07-14T09:51:08.777684Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = pd.DataFrame({'PassengerId': PassengerId, 'Survived':Y_pred })\nsubmission.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-14T09:51:08.780317Z","iopub.execute_input":"2022-07-14T09:51:08.780676Z","iopub.status.idle":"2022-07-14T09:51:08.796775Z","shell.execute_reply.started":"2022-07-14T09:51:08.780647Z","shell.execute_reply":"2022-07-14T09:51:08.795425Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.to_csv(\"Titanic_predictions_Voyager31.csv\", index = False)","metadata":{"execution":{"iopub.status.busy":"2022-07-14T09:53:31.495910Z","iopub.execute_input":"2022-07-14T09:53:31.496331Z","iopub.status.idle":"2022-07-14T09:53:31.505492Z","shell.execute_reply.started":"2022-07-14T09:53:31.496299Z","shell.execute_reply":"2022-07-14T09:53:31.504219Z"},"trusted":true},"execution_count":null,"outputs":[]}]}