{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"### Titanic - Machine Learning from Disaster","metadata":{}},{"cell_type":"markdown","source":"The goal of this project is to preprocess data and use Machine Learning algorithms to predict whether the passengers in the test dataset survived or not.\nWe have to csv files with us; __train.csv__ and __test.csv__.\nBasically, the plan is to clean both the datasets, remove nulls, and then create a machine learning model using __train__ dataset, and use that model to predict the survival of __test__.\n","metadata":{}},{"cell_type":"markdown","source":"| Variable      | Definition | Key     |\n| :---        |    :----:   |          ---: |\n| Survival      | Survival       | 0 = No, 1 = Yes   |\n| pclass   | Ticket class        | 1 = 1st, 2 = 2nd, 3 = 3rd    |\n|Sex  |Sex| |\n|Age|Age in years||\n|sibsp|# of siblings / spouses aboard the Titanic||\n|parch|# of parents / children aboard the Titanic||\n|ticket|Ticket number||\n|fare|Passenger fare||\n|cabin|Cabin number||\n|embarked|Port of Embarkation|C = Cherbourg, Q = Queenstown, S = Southampton|","metadata":{}},{"cell_type":"code","source":"#First step is to import the data analysis and visualization libraries","metadata":{"execution":{"iopub.status.busy":"2022-08-11T20:58:48.874981Z","iopub.execute_input":"2022-08-11T20:58:48.875497Z","iopub.status.idle":"2022-08-11T20:58:48.881491Z","shell.execute_reply.started":"2022-08-11T20:58:48.875454Z","shell.execute_reply":"2022-08-11T20:58:48.880094Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np","metadata":{"execution":{"iopub.status.busy":"2022-08-11T20:58:49.001141Z","iopub.execute_input":"2022-08-11T20:58:49.001598Z","iopub.status.idle":"2022-08-11T20:58:49.006305Z","shell.execute_reply.started":"2022-08-11T20:58:49.001563Z","shell.execute_reply":"2022-08-11T20:58:49.005348Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import seaborn as sns\nimport matplotlib.pyplot as plt","metadata":{"execution":{"iopub.status.busy":"2022-08-11T20:58:49.007931Z","iopub.execute_input":"2022-08-11T20:58:49.008519Z","iopub.status.idle":"2022-08-11T20:58:49.020933Z","shell.execute_reply.started":"2022-08-11T20:58:49.008484Z","shell.execute_reply":"2022-08-11T20:58:49.019897Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%matplotlib inline","metadata":{"execution":{"iopub.status.busy":"2022-08-11T20:58:49.023746Z","iopub.execute_input":"2022-08-11T20:58:49.024469Z","iopub.status.idle":"2022-08-11T20:58:49.034967Z","shell.execute_reply.started":"2022-08-11T20:58:49.024428Z","shell.execute_reply":"2022-08-11T20:58:49.033954Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.set_style('whitegrid')","metadata":{"execution":{"iopub.status.busy":"2022-08-11T20:58:49.037268Z","iopub.execute_input":"2022-08-11T20:58:49.037622Z","iopub.status.idle":"2022-08-11T20:58:49.048453Z","shell.execute_reply.started":"2022-08-11T20:58:49.037592Z","shell.execute_reply":"2022-08-11T20:58:49.047098Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv('../input/titanic/train.csv')\ntest = pd.read_csv('../input/titanic/test.csv')","metadata":{"execution":{"iopub.status.busy":"2022-08-11T20:58:49.049712Z","iopub.execute_input":"2022-08-11T20:58:49.050560Z","iopub.status.idle":"2022-08-11T20:58:49.083920Z","shell.execute_reply.started":"2022-08-11T20:58:49.050519Z","shell.execute_reply":"2022-08-11T20:58:49.082829Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Lets work with train dataset first","metadata":{}},{"cell_type":"code","source":"train.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-11T20:58:49.085165Z","iopub.execute_input":"2022-08-11T20:58:49.085956Z","iopub.status.idle":"2022-08-11T20:58:49.103802Z","shell.execute_reply.started":"2022-08-11T20:58:49.085918Z","shell.execute_reply":"2022-08-11T20:58:49.102632Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.info()","metadata":{"execution":{"iopub.status.busy":"2022-08-11T20:58:49.105464Z","iopub.execute_input":"2022-08-11T20:58:49.105814Z","iopub.status.idle":"2022-08-11T20:58:49.124401Z","shell.execute_reply.started":"2022-08-11T20:58:49.105781Z","shell.execute_reply":"2022-08-11T20:58:49.123228Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.describe()","metadata":{"scrolled":true,"execution":{"iopub.status.busy":"2022-08-11T20:58:49.126000Z","iopub.execute_input":"2022-08-11T20:58:49.126931Z","iopub.status.idle":"2022-08-11T20:58:49.167296Z","shell.execute_reply.started":"2022-08-11T20:58:49.126893Z","shell.execute_reply":"2022-08-11T20:58:49.165927Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We can see from *.info()* that columns Age, Cabin and Embarked contain nulls. We can show that visually, by creating a heatmap.         ","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize = (10,6))\nsns.heatmap(train.isnull(), cbar = False, yticklabels=False, cmap= 'viridis')","metadata":{"scrolled":true,"execution":{"iopub.status.busy":"2022-08-11T20:58:49.168816Z","iopub.execute_input":"2022-08-11T20:58:49.169572Z","iopub.status.idle":"2022-08-11T20:58:49.446156Z","shell.execute_reply.started":"2022-08-11T20:58:49.169535Z","shell.execute_reply":"2022-08-11T20:58:49.444947Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"####  Lets do some exploratory data analysis ","metadata":{}},{"cell_type":"code","source":"sns.catplot(x=\"Survived\", kind=\"count\", palette=\"coolwarm\", data=train, hue = 'Sex')","metadata":{"execution":{"iopub.status.busy":"2022-08-11T20:58:49.451657Z","iopub.execute_input":"2022-08-11T20:58:49.452067Z","iopub.status.idle":"2022-08-11T20:58:49.859686Z","shell.execute_reply.started":"2022-08-11T20:58:49.452033Z","shell.execute_reply":"2022-08-11T20:58:49.858311Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The barplot indicates that the survival rate was very high for females, whereas large majority of the males did not survive.","metadata":{}},{"cell_type":"code","source":"sns.catplot(x=\"Survived\", kind=\"count\", palette=\"coolwarm\", data=train, hue = 'Pclass')","metadata":{"execution":{"iopub.status.busy":"2022-08-11T20:58:49.861464Z","iopub.execute_input":"2022-08-11T20:58:49.862284Z","iopub.status.idle":"2022-08-11T20:58:50.298544Z","shell.execute_reply.started":"2022-08-11T20:58:49.862234Z","shell.execute_reply":"2022-08-11T20:58:50.297292Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Most of the deaths were passengers in the 3rd class,which was the lower class, while most survivors belonged to 1st class.","metadata":{}},{"cell_type":"code","source":"sns.countplot(train['SibSp'], palette='cool')","metadata":{"execution":{"iopub.status.busy":"2022-08-11T20:58:50.300451Z","iopub.execute_input":"2022-08-11T20:58:50.300829Z","iopub.status.idle":"2022-08-11T20:58:50.543518Z","shell.execute_reply.started":"2022-08-11T20:58:50.300795Z","shell.execute_reply":"2022-08-11T20:58:50.542158Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.pivot_table(index='SibSp', columns= 'Survived', aggfunc= len)['Age'].plot(kind = 'bar', figsize = (10,8))","metadata":{"execution":{"iopub.status.busy":"2022-08-11T20:58:50.545576Z","iopub.execute_input":"2022-08-11T20:58:50.546084Z","iopub.status.idle":"2022-08-11T20:58:50.897434Z","shell.execute_reply.started":"2022-08-11T20:58:50.546033Z","shell.execute_reply":"2022-08-11T20:58:50.895893Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Approximately only 1/3rd of the solo passengers survived, whereas about 60 percent of the passengers travelling with one other person, survived. These passengers could a married couple most likely. ","metadata":{}},{"cell_type":"code","source":"train['Fare'].hist(color='blue',bins=40,figsize=(10,6))","metadata":{"execution":{"iopub.status.busy":"2022-08-11T20:58:50.899372Z","iopub.execute_input":"2022-08-11T20:58:50.899896Z","iopub.status.idle":"2022-08-11T20:58:51.233775Z","shell.execute_reply.started":"2022-08-11T20:58:50.899849Z","shell.execute_reply":"2022-08-11T20:58:51.232480Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.countplot(train['Embarked'], hue = train['Survived'])","metadata":{"scrolled":true,"execution":{"iopub.status.busy":"2022-08-11T20:58:51.235533Z","iopub.execute_input":"2022-08-11T20:58:51.236065Z","iopub.status.idle":"2022-08-11T20:58:51.494441Z","shell.execute_reply.started":"2022-08-11T20:58:51.235998Z","shell.execute_reply":"2022-08-11T20:58:51.493097Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.distplot(train[train['Survived'] == 1]['Age'].dropna(),color= 'green', kde= False, bins = 30, hist_kws={'alpha': 0.7})\nsns.distplot(train[train['Survived'] == 0]['Age'].dropna(),color= 'red', kde= False, bins = 30, hist_kws={'alpha': 0.4})\n\nplt.xlim(0,80)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T20:58:51.496395Z","iopub.execute_input":"2022-08-11T20:58:51.496935Z","iopub.status.idle":"2022-08-11T20:58:51.884383Z","shell.execute_reply.started":"2022-08-11T20:58:51.496884Z","shell.execute_reply":"2022-08-11T20:58:51.883135Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(10,6))\nsns.boxplot(x = 'Pclass', y ='Age', data = train, hue ='Sex')","metadata":{"execution":{"iopub.status.busy":"2022-08-11T20:58:51.886424Z","iopub.execute_input":"2022-08-11T20:58:51.887066Z","iopub.status.idle":"2022-08-11T20:58:52.235733Z","shell.execute_reply.started":"2022-08-11T20:58:51.886999Z","shell.execute_reply":"2022-08-11T20:58:52.233919Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Handling missing data","metadata":{}},{"cell_type":"code","source":"#Check for amount of missing data in each column:","metadata":{"execution":{"iopub.status.busy":"2022-08-11T20:58:52.237447Z","iopub.execute_input":"2022-08-11T20:58:52.238669Z","iopub.status.idle":"2022-08-11T20:58:52.243270Z","shell.execute_reply.started":"2022-08-11T20:58:52.238621Z","shell.execute_reply":"2022-08-11T20:58:52.242004Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('% of missing data in Cabin column: ',   100*train['Cabin'].isnull().sum()/891)\nprint('\\n')\nprint('% of missing data in Age column: ',   100*train['Age'].isnull().sum()/891)\nprint('\\n')\nprint('% of missing data in Embarked column: ',   100*train['Embarked'].isnull().sum()/891)","metadata":{"scrolled":true,"execution":{"iopub.status.busy":"2022-08-11T20:58:52.244837Z","iopub.execute_input":"2022-08-11T20:58:52.245983Z","iopub.status.idle":"2022-08-11T20:58:52.260944Z","shell.execute_reply.started":"2022-08-11T20:58:52.245924Z","shell.execute_reply":"2022-08-11T20:58:52.259700Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The percentage of missing data in __Cabin__ column too large to actually be able to use it, therefore we will drop the column.","metadata":{}},{"cell_type":"code","source":"train.drop('Cabin', axis = 1, inplace = True)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T20:58:52.262889Z","iopub.execute_input":"2022-08-11T20:58:52.263700Z","iopub.status.idle":"2022-08-11T20:58:52.274413Z","shell.execute_reply.started":"2022-08-11T20:58:52.263604Z","shell.execute_reply":"2022-08-11T20:58:52.273113Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize = (10,6))\nsns.heatmap(train.isnull(), cbar = False, yticklabels=False, cmap= 'viridis')","metadata":{"execution":{"iopub.status.busy":"2022-08-11T20:58:52.276298Z","iopub.execute_input":"2022-08-11T20:58:52.276684Z","iopub.status.idle":"2022-08-11T20:58:52.542282Z","shell.execute_reply.started":"2022-08-11T20:58:52.276649Z","shell.execute_reply":"2022-08-11T20:58:52.540686Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Now we will use a function to fill missing values in the __Age__ column usning the __PClass__ and __Sex__ column.","metadata":{}},{"cell_type":"code","source":"grp = train.groupby(['Pclass','Sex'])[['Age']].mean().round().reset_index()","metadata":{"scrolled":true,"execution":{"iopub.status.busy":"2022-08-11T20:58:52.544095Z","iopub.execute_input":"2022-08-11T20:58:52.544632Z","iopub.status.idle":"2022-08-11T20:58:52.562442Z","shell.execute_reply.started":"2022-08-11T20:58:52.544594Z","shell.execute_reply":"2022-08-11T20:58:52.560716Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def fill_Age(x):\n    return grp[(grp.Pclass ==x.Pclass) &(grp.Sex==x.Sex)]['Age'].values[0]","metadata":{"execution":{"iopub.status.busy":"2022-08-11T20:58:52.564618Z","iopub.execute_input":"2022-08-11T20:58:52.564987Z","iopub.status.idle":"2022-08-11T20:58:52.571450Z","shell.execute_reply.started":"2022-08-11T20:58:52.564954Z","shell.execute_reply":"2022-08-11T20:58:52.570087Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# def impute_age(cols):\n#     Age = cols[0]\n#     Pclass = cols[1]\n    \n#     if pd.isnull(Age):\n#         if Pclass ==1:\n#             return  38\n#         elif Pclass ==2:\n#             return 30\n#         else:\n#             return 25\n#     else:\n#         return Age","metadata":{"execution":{"iopub.status.busy":"2022-08-11T20:58:52.573234Z","iopub.execute_input":"2022-08-11T20:58:52.573701Z","iopub.status.idle":"2022-08-11T20:58:52.584574Z","shell.execute_reply.started":"2022-08-11T20:58:52.573665Z","shell.execute_reply":"2022-08-11T20:58:52.583337Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train['Age'] = train[['Age', 'Pclass']].apply(impute_age, axis = 1)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T20:58:52.586246Z","iopub.execute_input":"2022-08-11T20:58:52.586900Z","iopub.status.idle":"2022-08-11T20:58:52.597555Z","shell.execute_reply.started":"2022-08-11T20:58:52.586848Z","shell.execute_reply":"2022-08-11T20:58:52.596267Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['Age'] = train.apply(lambda x: fill_Age(x) if np.isnan(x['Age']) else x['Age'], axis = 1)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T20:58:52.598634Z","iopub.execute_input":"2022-08-11T20:58:52.598980Z","iopub.status.idle":"2022-08-11T20:58:52.790466Z","shell.execute_reply.started":"2022-08-11T20:58:52.598948Z","shell.execute_reply":"2022-08-11T20:58:52.789123Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Since there are only 2 columns in __Embarked__ that are missing values, we can drop those rows from the train datframe.","metadata":{}},{"cell_type":"code","source":"train.dropna(inplace = True)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T20:58:52.791908Z","iopub.execute_input":"2022-08-11T20:58:52.792344Z","iopub.status.idle":"2022-08-11T20:58:52.801897Z","shell.execute_reply.started":"2022-08-11T20:58:52.792309Z","shell.execute_reply":"2022-08-11T20:58:52.800591Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize = (10,6))\nsns.heatmap(train.isnull(), cbar = False, yticklabels=False, cmap= 'viridis')","metadata":{"execution":{"iopub.status.busy":"2022-08-11T20:58:52.813257Z","iopub.execute_input":"2022-08-11T20:58:52.813690Z","iopub.status.idle":"2022-08-11T20:58:53.075380Z","shell.execute_reply.started":"2022-08-11T20:58:52.813658Z","shell.execute_reply":"2022-08-11T20:58:53.074113Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The train dataframe is free from nulls, and can be used to make predictions.","metadata":{}},{"cell_type":"code","source":"train.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-11T20:58:53.077186Z","iopub.execute_input":"2022-08-11T20:58:53.078342Z","iopub.status.idle":"2022-08-11T20:58:53.095424Z","shell.execute_reply.started":"2022-08-11T20:58:53.078293Z","shell.execute_reply":"2022-08-11T20:58:53.094506Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = train.copy()","metadata":{"execution":{"iopub.status.busy":"2022-08-11T20:58:53.096555Z","iopub.execute_input":"2022-08-11T20:58:53.097717Z","iopub.status.idle":"2022-08-11T20:58:53.103159Z","shell.execute_reply.started":"2022-08-11T20:58:53.097655Z","shell.execute_reply":"2022-08-11T20:58:53.101733Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We are going to implement *.get_dummies()* method to encode categorical features into the Machine learning models.","metadata":{}},{"cell_type":"code","source":"sex = pd.get_dummies(train_df['Sex'], drop_first=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T20:58:53.104738Z","iopub.execute_input":"2022-08-11T20:58:53.105084Z","iopub.status.idle":"2022-08-11T20:58:53.118374Z","shell.execute_reply.started":"2022-08-11T20:58:53.105050Z","shell.execute_reply":"2022-08-11T20:58:53.116662Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"embarked = pd.get_dummies(train_df['Embarked'], drop_first=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T20:58:53.120343Z","iopub.execute_input":"2022-08-11T20:58:53.120748Z","iopub.status.idle":"2022-08-11T20:58:53.133154Z","shell.execute_reply.started":"2022-08-11T20:58:53.120711Z","shell.execute_reply":"2022-08-11T20:58:53.131766Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"To avoid overfitting the model, we can check who was travelling alone and who was not, and create a column based on that information, since it doesnt help us much to see who was travelling with theire siblings/spouse or parents, and how many of them were there. So we'll create a new column __IsAlone__ to ease the complexity of these two columns:","metadata":{}},{"cell_type":"code","source":"train_df['IsAlone'] = np.where((train_df['SibSp'] == 0) & (train_df['Parch'] == 0), 0,1)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T20:58:53.134914Z","iopub.execute_input":"2022-08-11T20:58:53.136546Z","iopub.status.idle":"2022-08-11T20:58:53.147135Z","shell.execute_reply.started":"2022-08-11T20:58:53.136487Z","shell.execute_reply":"2022-08-11T20:58:53.146077Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.drop(['PassengerId', 'Name', 'Sex', 'Ticket','Embarked'],axis = 1, inplace = True)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T20:58:53.148783Z","iopub.execute_input":"2022-08-11T20:58:53.149629Z","iopub.status.idle":"2022-08-11T20:58:53.158333Z","shell.execute_reply.started":"2022-08-11T20:58:53.149577Z","shell.execute_reply":"2022-08-11T20:58:53.156974Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.drop(['SibSp', 'Parch'],axis = 1, inplace = True)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T20:58:53.159705Z","iopub.execute_input":"2022-08-11T20:58:53.160303Z","iopub.status.idle":"2022-08-11T20:58:53.171063Z","shell.execute_reply.started":"2022-08-11T20:58:53.160264Z","shell.execute_reply":"2022-08-11T20:58:53.169877Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.concat([train_df,sex, embarked], axis = 1)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T20:58:53.172690Z","iopub.execute_input":"2022-08-11T20:58:53.173450Z","iopub.status.idle":"2022-08-11T20:58:53.183121Z","shell.execute_reply.started":"2022-08-11T20:58:53.173410Z","shell.execute_reply":"2022-08-11T20:58:53.182048Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Our final dataframe looks like this:","metadata":{}},{"cell_type":"code","source":"train_df","metadata":{"scrolled":true,"execution":{"iopub.status.busy":"2022-08-11T20:58:53.184948Z","iopub.execute_input":"2022-08-11T20:58:53.185711Z","iopub.status.idle":"2022-08-11T20:58:53.208472Z","shell.execute_reply.started":"2022-08-11T20:58:53.185663Z","shell.execute_reply":"2022-08-11T20:58:53.207333Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Lets try running our Classification models on this dataset only, to see which model gives us the best results.\nFor that,we will split our data into train test split, feed our train data to the model, and predict our test data.","metadata":{}},{"cell_type":"code","source":"X = train_df.drop('Survived', axis =1)\ny = train_df['Survived']","metadata":{"execution":{"iopub.status.busy":"2022-08-11T20:58:53.210001Z","iopub.execute_input":"2022-08-11T20:58:53.210591Z","iopub.status.idle":"2022-08-11T20:58:53.217796Z","shell.execute_reply.started":"2022-08-11T20:58:53.210551Z","shell.execute_reply":"2022-08-11T20:58:53.216680Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split","metadata":{"execution":{"iopub.status.busy":"2022-08-11T20:58:53.219200Z","iopub.execute_input":"2022-08-11T20:58:53.219574Z","iopub.status.idle":"2022-08-11T20:58:53.230085Z","shell.execute_reply.started":"2022-08-11T20:58:53.219542Z","shell.execute_reply":"2022-08-11T20:58:53.228478Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.3, random_state=101)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T20:58:53.231581Z","iopub.execute_input":"2022-08-11T20:58:53.232416Z","iopub.status.idle":"2022-08-11T20:58:53.246715Z","shell.execute_reply.started":"2022-08-11T20:58:53.232362Z","shell.execute_reply":"2022-08-11T20:58:53.245321Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Importing all the classification models we are going to fit our data into:\n\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.tree import DecisionTreeClassifier\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.neighbors import KNeighborsClassifier\nfrom sklearn.svm import SVC\nfrom sklearn.model_selection import GridSearchCV\nfrom sklearn.naive_bayes import MultinomialNB\nfrom sklearn.metrics import confusion_matrix, classification_report, accuracy_score,roc_auc_score","metadata":{"execution":{"iopub.status.busy":"2022-08-11T20:58:53.248678Z","iopub.execute_input":"2022-08-11T20:58:53.249135Z","iopub.status.idle":"2022-08-11T20:58:53.260261Z","shell.execute_reply.started":"2022-08-11T20:58:53.249094Z","shell.execute_reply":"2022-08-11T20:58:53.258270Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Logistic Regression Classifier\n\nlog_model = LogisticRegression(max_iter= 1000)\n\nlog_model.fit(X_train,y_train)\nlog_pred = log_model.predict(X_test)\n\nprint(confusion_matrix(y_test, log_pred))\nprint('\\n')\nprint(classification_report(y_test, log_pred))\nprint('\\n')\nprint(accuracy_score(y_test,log_pred))\nprint('\\n')\nprint(roc_auc_score(y_test, log_pred))","metadata":{"execution":{"iopub.status.busy":"2022-08-11T20:58:53.262874Z","iopub.execute_input":"2022-08-11T20:58:53.263516Z","iopub.status.idle":"2022-08-11T20:58:53.335402Z","shell.execute_reply.started":"2022-08-11T20:58:53.263461Z","shell.execute_reply":"2022-08-11T20:58:53.333922Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Decision Tree Classifier\n\ndtree = DecisionTreeClassifier()\n\ndtree.fit(X_train,y_train)\ndt_pred = dtree.predict(X_test)\n\nprint(confusion_matrix(y_test, dt_pred))\nprint('\\n')\nprint(classification_report(y_test, dt_pred))\nprint('\\n')\nprint(accuracy_score(y_test,dt_pred))\nprint('\\n')\nprint(roc_auc_score(y_test, dt_pred))","metadata":{"execution":{"iopub.status.busy":"2022-08-11T20:58:53.337088Z","iopub.execute_input":"2022-08-11T20:58:53.337496Z","iopub.status.idle":"2022-08-11T20:58:53.361953Z","shell.execute_reply.started":"2022-08-11T20:58:53.337461Z","shell.execute_reply":"2022-08-11T20:58:53.360495Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#K-Nearest Neighbours Classifier\n\nknn = KNeighborsClassifier(n_neighbors=2)\n\nknn.fit(X_train,y_train)\nknn_pred = knn.predict(X_test)\n\nprint(confusion_matrix(y_test, knn_pred))\nprint('\\n')\nprint(classification_report(y_test, knn_pred))\nprint('\\n')\nprint(accuracy_score(y_test,knn_pred))\nprint('\\n')\nprint(roc_auc_score(y_test, knn_pred))","metadata":{"execution":{"iopub.status.busy":"2022-08-11T20:58:53.364233Z","iopub.execute_input":"2022-08-11T20:58:53.364610Z","iopub.status.idle":"2022-08-11T20:58:53.398277Z","shell.execute_reply.started":"2022-08-11T20:58:53.364578Z","shell.execute_reply":"2022-08-11T20:58:53.396500Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Random Forest Classifier\n\n\nrfc = RandomForestClassifier(verbose=3)\n\nrfc.fit(X_train,y_train)\nrfc_pred = rfc.predict(X_test)\n\nprint(confusion_matrix(y_test, rfc_pred))\nprint('\\n')\nprint(classification_report(y_test, rfc_pred))\nprint('\\n')\nprint(accuracy_score(y_test,rfc_pred))\nprint('\\n')\nprint(roc_auc_score(y_test, rfc_pred))","metadata":{"execution":{"iopub.status.busy":"2022-08-11T20:58:53.400034Z","iopub.execute_input":"2022-08-11T20:58:53.400390Z","iopub.status.idle":"2022-08-11T20:58:53.649384Z","shell.execute_reply.started":"2022-08-11T20:58:53.400358Z","shell.execute_reply":"2022-08-11T20:58:53.647756Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Support Vector Machines classifier using Grid Search\n\nparam_grid = {'C':[0.1, 1,10,100,1000], 'gamma': [1,0.1,0.01,0.001,0.0001]}\ngrid = GridSearchCV(SVC(),param_grid, verbose=3)\n\ngrid.fit(X_train,y_train)\ngrid_pred = grid.predict(X_test)\n\nprint(confusion_matrix(y_test, grid_pred))\nprint('\\n')\nprint(classification_report(y_test, grid_pred))\nprint('\\n')\nprint(accuracy_score(y_test,grid_pred))\nprint('\\n')\nprint(roc_auc_score(y_test, grid_pred))","metadata":{"execution":{"iopub.status.busy":"2022-08-11T20:58:53.650967Z","iopub.execute_input":"2022-08-11T20:58:53.651616Z","iopub.status.idle":"2022-08-11T20:58:57.634554Z","shell.execute_reply.started":"2022-08-11T20:58:53.651564Z","shell.execute_reply":"2022-08-11T20:58:57.633282Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Naive Bayes Classifier\n\nnb = MultinomialNB()\n\nnb.fit(X_train,y_train)\nnb_pred = nb.predict(X_test)\n\nprint(confusion_matrix(y_test, nb_pred))\nprint('\\n')\nprint(classification_report(y_test, nb_pred))\nprint('\\n')\nprint(accuracy_score(y_test,nb_pred))\nprint('\\n')\nprint(roc_auc_score(y_test, nb_pred))","metadata":{"execution":{"iopub.status.busy":"2022-08-11T20:58:57.635793Z","iopub.execute_input":"2022-08-11T20:58:57.636518Z","iopub.status.idle":"2022-08-11T20:58:57.657687Z","shell.execute_reply.started":"2022-08-11T20:58:57.636483Z","shell.execute_reply":"2022-08-11T20:58:57.656562Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"models = ['log_model','dtree','knn','rfc','grid','nb']\nscore = [accuracy_score(y_test,log_pred), accuracy_score(y_test,dt_pred), accuracy_score(y_test,knn_pred),\n        accuracy_score(y_test,rfc_pred), accuracy_score(y_test,grid_pred), accuracy_score(y_test,nb_pred)]","metadata":{"execution":{"iopub.status.busy":"2022-08-11T20:58:57.659984Z","iopub.execute_input":"2022-08-11T20:58:57.660354Z","iopub.status.idle":"2022-08-11T20:58:57.670359Z","shell.execute_reply.started":"2022-08-11T20:58:57.660322Z","shell.execute_reply":"2022-08-11T20:58:57.669244Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"scores = pd.DataFrame({'Model' : models,'Score' : score})","metadata":{"execution":{"iopub.status.busy":"2022-08-11T20:58:57.671957Z","iopub.execute_input":"2022-08-11T20:58:57.672369Z","iopub.status.idle":"2022-08-11T20:58:57.680522Z","shell.execute_reply.started":"2022-08-11T20:58:57.672335Z","shell.execute_reply":"2022-08-11T20:58:57.679227Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#We can show our accuracy scores as a dataframe:\n\nscores.sort_values(by='Score', ascending=False)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T20:58:57.681916Z","iopub.execute_input":"2022-08-11T20:58:57.682667Z","iopub.status.idle":"2022-08-11T20:58:57.703376Z","shell.execute_reply.started":"2022-08-11T20:58:57.682623Z","shell.execute_reply":"2022-08-11T20:58:57.702068Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The best predictions were made by __SVC classifier using Grid Search__. Therefore for predicting the values in the test.csv file, we will use only that model.","metadata":{}},{"cell_type":"markdown","source":"Let us check our test data now, and clean it before predicting our survivors.","metadata":{}},{"cell_type":"code","source":"test_df = test.copy()","metadata":{"execution":{"iopub.status.busy":"2022-08-11T20:58:57.705047Z","iopub.execute_input":"2022-08-11T20:58:57.706240Z","iopub.status.idle":"2022-08-11T20:58:57.712371Z","shell.execute_reply.started":"2022-08-11T20:58:57.706178Z","shell.execute_reply":"2022-08-11T20:58:57.711033Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize = (10,6))\nsns.heatmap(test.isnull(), cbar = False, yticklabels=False, cmap= 'viridis')","metadata":{"execution":{"iopub.status.busy":"2022-08-11T20:58:57.714147Z","iopub.execute_input":"2022-08-11T20:58:57.714725Z","iopub.status.idle":"2022-08-11T20:58:57.955278Z","shell.execute_reply.started":"2022-08-11T20:58:57.714688Z","shell.execute_reply":"2022-08-11T20:58:57.954095Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.info()","metadata":{"execution":{"iopub.status.busy":"2022-08-11T20:58:57.957134Z","iopub.execute_input":"2022-08-11T20:58:57.957617Z","iopub.status.idle":"2022-08-11T20:58:57.974734Z","shell.execute_reply.started":"2022-08-11T20:58:57.957569Z","shell.execute_reply":"2022-08-11T20:58:57.973390Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.drop('Cabin', axis = 1, inplace = True)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T20:58:57.976843Z","iopub.execute_input":"2022-08-11T20:58:57.977364Z","iopub.status.idle":"2022-08-11T20:58:57.989944Z","shell.execute_reply.started":"2022-08-11T20:58:57.977317Z","shell.execute_reply":"2022-08-11T20:58:57.989050Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test[test['Fare'].isnull()]","metadata":{"execution":{"iopub.status.busy":"2022-08-11T20:58:57.991376Z","iopub.execute_input":"2022-08-11T20:58:57.992286Z","iopub.status.idle":"2022-08-11T20:58:58.013703Z","shell.execute_reply.started":"2022-08-11T20:58:57.992247Z","shell.execute_reply":"2022-08-11T20:58:58.012544Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Since there is only one value missing for Fare column, we will take a mean Fare using other columns as reference.\n\ntest[(test['Pclass'] == 3) & (test['Embarked'] == 'S') &(test['Parch']==0) &(test['SibSp']==0)]['Fare'].mean()","metadata":{"execution":{"iopub.status.busy":"2022-08-11T20:58:58.015094Z","iopub.execute_input":"2022-08-11T20:58:58.015632Z","iopub.status.idle":"2022-08-11T20:58:58.025323Z","shell.execute_reply.started":"2022-08-11T20:58:58.015596Z","shell.execute_reply":"2022-08-11T20:58:58.024172Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.loc[152,'Fare'] = 9.34","metadata":{"execution":{"iopub.status.busy":"2022-08-11T20:58:58.027835Z","iopub.execute_input":"2022-08-11T20:58:58.028885Z","iopub.status.idle":"2022-08-11T20:58:58.037315Z","shell.execute_reply.started":"2022-08-11T20:58:58.028836Z","shell.execute_reply":"2022-08-11T20:58:58.036085Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# For the Age column, we are going to use the same imputation function that we used before.\n\ntest['Age'] = test.apply(lambda x: fill_Age(x) if np.isnan(x['Age']) else x['Age'], axis = 1)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T20:58:58.038796Z","iopub.execute_input":"2022-08-11T20:58:58.039372Z","iopub.status.idle":"2022-08-11T20:58:58.121829Z","shell.execute_reply.started":"2022-08-11T20:58:58.039338Z","shell.execute_reply":"2022-08-11T20:58:58.120902Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize = (10,6))\nsns.heatmap(test.isnull(), cbar = False, yticklabels=False, cmap= 'viridis')","metadata":{"execution":{"iopub.status.busy":"2022-08-11T20:58:58.123114Z","iopub.execute_input":"2022-08-11T20:58:58.123649Z","iopub.status.idle":"2022-08-11T20:58:58.351662Z","shell.execute_reply.started":"2022-08-11T20:58:58.123616Z","shell.execute_reply":"2022-08-11T20:58:58.350606Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Our test data is clean now, and the final predictions can begin.","metadata":{}},{"cell_type":"code","source":"X_TRAIN = X","metadata":{"execution":{"iopub.status.busy":"2022-08-11T20:58:58.355282Z","iopub.execute_input":"2022-08-11T20:58:58.355634Z","iopub.status.idle":"2022-08-11T20:58:58.360747Z","shell.execute_reply.started":"2022-08-11T20:58:58.355603Z","shell.execute_reply":"2022-08-11T20:58:58.359564Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_TRAIN.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-11T20:58:58.362584Z","iopub.execute_input":"2022-08-11T20:58:58.363376Z","iopub.status.idle":"2022-08-11T20:58:58.380983Z","shell.execute_reply.started":"2022-08-11T20:58:58.363328Z","shell.execute_reply":"2022-08-11T20:58:58.379816Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Y_TRAIN = y","metadata":{"execution":{"iopub.status.busy":"2022-08-11T20:58:58.383988Z","iopub.execute_input":"2022-08-11T20:58:58.384529Z","iopub.status.idle":"2022-08-11T20:58:58.396538Z","shell.execute_reply.started":"2022-08-11T20:58:58.384495Z","shell.execute_reply":"2022-08-11T20:58:58.395479Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-11T20:58:58.398182Z","iopub.execute_input":"2022-08-11T20:58:58.398946Z","iopub.status.idle":"2022-08-11T20:58:58.423432Z","shell.execute_reply.started":"2022-08-11T20:58:58.398896Z","shell.execute_reply":"2022-08-11T20:58:58.422331Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_sex = pd.get_dummies(test['Sex'], drop_first=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T20:58:58.424971Z","iopub.execute_input":"2022-08-11T20:58:58.425831Z","iopub.status.idle":"2022-08-11T20:58:58.435831Z","shell.execute_reply.started":"2022-08-11T20:58:58.425792Z","shell.execute_reply":"2022-08-11T20:58:58.434736Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_embarked = pd.get_dummies(test['Embarked'], drop_first=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T20:58:58.437215Z","iopub.execute_input":"2022-08-11T20:58:58.438382Z","iopub.status.idle":"2022-08-11T20:58:58.450024Z","shell.execute_reply.started":"2022-08-11T20:58:58.438337Z","shell.execute_reply":"2022-08-11T20:58:58.449169Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test['IsAlone'] = np.where((test['SibSp'] == 0) & (test['Parch'] == 0), 0,1)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T20:58:58.451264Z","iopub.execute_input":"2022-08-11T20:58:58.451758Z","iopub.status.idle":"2022-08-11T20:58:58.462691Z","shell.execute_reply.started":"2022-08-11T20:58:58.451727Z","shell.execute_reply":"2022-08-11T20:58:58.461426Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-11T20:58:58.464391Z","iopub.execute_input":"2022-08-11T20:58:58.464942Z","iopub.status.idle":"2022-08-11T20:58:58.485590Z","shell.execute_reply.started":"2022-08-11T20:58:58.464907Z","shell.execute_reply":"2022-08-11T20:58:58.484705Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.drop(['PassengerId','Name', 'Sex', 'Ticket', 'Embarked','SibSp', 'Parch'], axis =1 , inplace = True)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T20:58:58.487141Z","iopub.execute_input":"2022-08-11T20:58:58.487464Z","iopub.status.idle":"2022-08-11T20:58:58.494623Z","shell.execute_reply.started":"2022-08-11T20:58:58.487435Z","shell.execute_reply":"2022-08-11T20:58:58.493412Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_TEST = pd.concat([test,test_embarked,test_sex], axis =1)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T20:58:58.495794Z","iopub.execute_input":"2022-08-11T20:58:58.496696Z","iopub.status.idle":"2022-08-11T20:58:58.509159Z","shell.execute_reply.started":"2022-08-11T20:58:58.496657Z","shell.execute_reply":"2022-08-11T20:58:58.507859Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_TEST.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-11T20:58:58.510694Z","iopub.execute_input":"2022-08-11T20:58:58.511662Z","iopub.status.idle":"2022-08-11T20:58:58.530448Z","shell.execute_reply.started":"2022-08-11T20:58:58.511620Z","shell.execute_reply":"2022-08-11T20:58:58.529052Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"grid.fit(X_TRAIN,Y_TRAIN)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T20:58:58.534115Z","iopub.execute_input":"2022-08-11T20:58:58.535371Z","iopub.status.idle":"2022-08-11T20:59:06.169613Z","shell.execute_reply.started":"2022-08-11T20:58:58.535319Z","shell.execute_reply":"2022-08-11T20:59:06.168389Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"GRID_Y = grid.predict(X_TEST)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T20:59:06.171956Z","iopub.execute_input":"2022-08-11T20:59:06.172370Z","iopub.status.idle":"2022-08-11T20:59:06.193409Z","shell.execute_reply.started":"2022-08-11T20:59:06.172334Z","shell.execute_reply":"2022-08-11T20:59:06.192035Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"GRID_Y","metadata":{"execution":{"iopub.status.busy":"2022-08-11T20:59:06.197410Z","iopub.execute_input":"2022-08-11T20:59:06.198170Z","iopub.status.idle":"2022-08-11T20:59:06.207215Z","shell.execute_reply.started":"2022-08-11T20:59:06.198129Z","shell.execute_reply":"2022-08-11T20:59:06.205934Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"result= pd.DataFrame({'PassengerId':test_df['PassengerId'], 'Survived': GRID_Y })","metadata":{"execution":{"iopub.status.busy":"2022-08-11T20:59:06.209117Z","iopub.execute_input":"2022-08-11T20:59:06.209854Z","iopub.status.idle":"2022-08-11T20:59:06.219169Z","shell.execute_reply.started":"2022-08-11T20:59:06.209807Z","shell.execute_reply":"2022-08-11T20:59:06.217842Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"result","metadata":{"execution":{"iopub.status.busy":"2022-08-11T20:59:06.220793Z","iopub.execute_input":"2022-08-11T20:59:06.221663Z","iopub.status.idle":"2022-08-11T20:59:06.241513Z","shell.execute_reply.started":"2022-08-11T20:59:06.221619Z","shell.execute_reply":"2022-08-11T20:59:06.240472Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"result['Survived'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-08-11T20:59:06.243414Z","iopub.execute_input":"2022-08-11T20:59:06.243896Z","iopub.status.idle":"2022-08-11T20:59:06.253851Z","shell.execute_reply.started":"2022-08-11T20:59:06.243846Z","shell.execute_reply":"2022-08-11T20:59:06.252589Z"},"trusted":true},"execution_count":null,"outputs":[]}]}