{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"### Import libraries","metadata":{}},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"import numpy as np \nimport pandas as pd\n\nimport matplotlib.pyplot as plt \n%matplotlib inline\nimport seaborn as sns\nsns.set(style=\"darkgrid\")\n\nimport warnings\nwarnings.filterwarnings('ignore')","metadata":{"execution":{"iopub.status.busy":"2022-07-16T05:17:32.615001Z","iopub.execute_input":"2022-07-16T05:17:32.615411Z","iopub.status.idle":"2022-07-16T05:17:33.237473Z","shell.execute_reply.started":"2022-07-16T05:17:32.615330Z","shell.execute_reply":"2022-07-16T05:17:33.236484Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# machine learning\nfrom sklearn.preprocessing import OneHotEncoder, LabelEncoder, StandardScaler\nfrom sklearn.metrics import roc_curve, auc\nfrom sklearn.model_selection import StratifiedKFold\n\n##\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.svm import SVC, LinearSVC\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.neighbors import KNeighborsClassifier\nfrom sklearn.naive_bayes import GaussianNB\nfrom sklearn.linear_model import Perceptron\nfrom sklearn.linear_model import SGDClassifier\nfrom sklearn.tree import DecisionTreeClassifier","metadata":{"execution":{"iopub.status.busy":"2022-07-16T05:17:45.440765Z","iopub.execute_input":"2022-07-16T05:17:45.441099Z","iopub.status.idle":"2022-07-16T05:17:45.738288Z","shell.execute_reply.started":"2022-07-16T05:17:45.441050Z","shell.execute_reply":"2022-07-16T05:17:45.737037Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"titanic_train = pd.read_csv(\"../input/titanic/train.csv\")\ntitanic_test = pd.read_csv(\"../input/titanic/test.csv\")\ndata_all = pd.concat([titanic_train, titanic_test], sort=True).reset_index(drop=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-16T05:19:54.837222Z","iopub.execute_input":"2022-07-16T05:19:54.838093Z","iopub.status.idle":"2022-07-16T05:19:54.880882Z","shell.execute_reply.started":"2022-07-16T05:19:54.838001Z","shell.execute_reply":"2022-07-16T05:19:54.879854Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Analyze and Inspecting DataSet\n\n#### Overview\n- PassengerId is the unique id\n- Survived\n    - 1 = Survived\n    - 0 = Not Survived\n- Pclass\n    - 1 = Upper Class\n    - 2 = Middle Class\n    - 3 = Lower Class\n- SibSp is the total number of the passengers' siblings and spouse\n- Parch is the total number of the passengers' parents and children\n- Ticket is the ticket number of the passenger\n- Fare is the passenger fare\n- Cabin is the cabin number of the passenger\n- Embarked is port of embarkation and it is a categorical feature which has 3 unique values (C, Q or S):\n    - C = Cherbourg\n    - Q = Queenstown\n    - S = Southampton","metadata":{}},{"cell_type":"code","source":"print(titanic_train.info())\ntitanic_train.sample(5)","metadata":{"execution":{"iopub.status.busy":"2022-07-16T05:22:17.606015Z","iopub.execute_input":"2022-07-16T05:22:17.606404Z","iopub.status.idle":"2022-07-16T05:22:17.647528Z","shell.execute_reply.started":"2022-07-16T05:22:17.606379Z","shell.execute_reply":"2022-07-16T05:22:17.646388Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(titanic_test.info())\ntitanic_test.sample(5)","metadata":{"execution":{"iopub.status.busy":"2022-07-16T05:22:29.605577Z","iopub.execute_input":"2022-07-16T05:22:29.605976Z","iopub.status.idle":"2022-07-16T05:22:29.635088Z","shell.execute_reply.started":"2022-07-16T05:22:29.605944Z","shell.execute_reply":"2022-07-16T05:22:29.634228Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"titanic_test.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-16T05:22:43.254960Z","iopub.execute_input":"2022-07-16T05:22:43.255350Z","iopub.status.idle":"2022-07-16T05:22:43.262796Z","shell.execute_reply.started":"2022-07-16T05:22:43.255321Z","shell.execute_reply":"2022-07-16T05:22:43.261586Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"titanic_train.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-16T05:22:48.638008Z","iopub.execute_input":"2022-07-16T05:22:48.638505Z","iopub.status.idle":"2022-07-16T05:22:48.646627Z","shell.execute_reply.started":"2022-07-16T05:22:48.638466Z","shell.execute_reply":"2022-07-16T05:22:48.645578Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_all.describe()","metadata":{"execution":{"iopub.status.busy":"2022-07-16T05:23:02.375193Z","iopub.execute_input":"2022-07-16T05:23:02.375587Z","iopub.status.idle":"2022-07-16T05:23:02.411382Z","shell.execute_reply.started":"2022-07-16T05:23:02.375558Z","shell.execute_reply":"2022-07-16T05:23:02.410312Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"titanic_train.describe(include=['O'])","metadata":{"execution":{"iopub.status.busy":"2022-07-16T05:23:09.391121Z","iopub.execute_input":"2022-07-16T05:23:09.391475Z","iopub.status.idle":"2022-07-16T05:23:09.414754Z","shell.execute_reply.started":"2022-07-16T05:23:09.391447Z","shell.execute_reply":"2022-07-16T05:23:09.414132Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"What is the distribution of categorical features?\n\n- Names are almost unique -> (count=unique=891)\n- Sex has two values 65% male -> (top=male, freq=577/count=891).\n- Cabin have a lot missig values.\n- Embarked takes three possible values -> S port used by most passengers (top=S)\n- Ticket feature has high ratio (22%) of duplicate values (unique=681)","metadata":{}},{"cell_type":"markdown","source":"## Missing Values","metadata":{}},{"cell_type":"code","source":"total = titanic_train.isnull().sum().sort_values(ascending=False)\nprecent = (titanic_train.isnull().sum() / titanic_train.isnull().count()).sort_values(ascending = False)\npd.concat([total , precent ] , keys = ['Total','Precent'] ,axis = 1).round(3)","metadata":{"execution":{"iopub.status.busy":"2022-07-16T05:24:11.751001Z","iopub.execute_input":"2022-07-16T05:24:11.751375Z","iopub.status.idle":"2022-07-16T05:24:11.775188Z","shell.execute_reply.started":"2022-07-16T05:24:11.751347Z","shell.execute_reply":"2022-07-16T05:24:11.773973Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(15,10))\nsns.heatmap(titanic_train.isna())","metadata":{"execution":{"iopub.status.busy":"2022-07-16T05:24:22.973982Z","iopub.execute_input":"2022-07-16T05:24:22.974344Z","iopub.status.idle":"2022-07-16T05:24:23.537838Z","shell.execute_reply.started":"2022-07-16T05:24:22.974316Z","shell.execute_reply":"2022-07-16T05:24:23.537013Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"total = titanic_test.isnull().sum().sort_values(ascending=False)\nprecent = (titanic_test.isnull().sum() / titanic_test.isnull().count()).sort_values(ascending = False)\npd.concat([total , precent ] , keys = ['Total','Precent'] ,axis = 1).round(3)","metadata":{"execution":{"iopub.status.busy":"2022-07-16T05:24:45.088166Z","iopub.execute_input":"2022-07-16T05:24:45.088515Z","iopub.status.idle":"2022-07-16T05:24:45.110564Z","shell.execute_reply.started":"2022-07-16T05:24:45.088487Z","shell.execute_reply":"2022-07-16T05:24:45.109503Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 1.1 Age","metadata":{}},{"cell_type":"code","source":"data_all.sample(5)","metadata":{"execution":{"iopub.status.busy":"2022-07-16T05:25:30.300502Z","iopub.execute_input":"2022-07-16T05:25:30.300835Z","iopub.status.idle":"2022-07-16T05:25:30.320840Z","shell.execute_reply.started":"2022-07-16T05:25:30.300807Z","shell.execute_reply":"2022-07-16T05:25:30.319693Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_all_corr = data_all.corr().abs().unstack().sort_values(kind=\"quicksort\", ascending=False).reset_index()\ndata_all_corr.rename(columns={'level_0' : 'Feature 1',\n                   'level_1' : 'Feature 2',\n                    0 : 'Correlation Coefficient'},inplace= True)\ndata_all_corr[data_all_corr['Feature 1'] == 'Age']","metadata":{"execution":{"iopub.status.busy":"2022-07-16T05:25:55.385204Z","iopub.execute_input":"2022-07-16T05:25:55.385569Z","iopub.status.idle":"2022-07-16T05:25:55.407938Z","shell.execute_reply.started":"2022-07-16T05:25:55.385538Z","shell.execute_reply":"2022-07-16T05:25:55.406820Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_all[['Pclass','Survived']].groupby('Pclass', as_index = False).mean()","metadata":{"execution":{"iopub.status.busy":"2022-07-16T05:26:07.183385Z","iopub.execute_input":"2022-07-16T05:26:07.183735Z","iopub.status.idle":"2022-07-16T05:26:07.199038Z","shell.execute_reply.started":"2022-07-16T05:26:07.183707Z","shell.execute_reply":"2022-07-16T05:26:07.198233Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# grid = sns.FacetGrid(train_df, col='Pclass', hue='Survived')\ngrid = sns.FacetGrid(data_all, col='Survived', row='Pclass', size=2.2, aspect=1.6)\ngrid.map(plt.hist, 'Age', alpha=.5, bins=20)\ngrid.add_legend();","metadata":{"execution":{"iopub.status.busy":"2022-07-16T05:26:17.808046Z","iopub.execute_input":"2022-07-16T05:26:17.808391Z","iopub.status.idle":"2022-07-16T05:26:19.555200Z","shell.execute_reply.started":"2022-07-16T05:26:17.808367Z","shell.execute_reply":"2022-07-16T05:26:19.554079Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- Missing values in Age are filled with median age, but using median age of the whole data set is not a good choice. Median age of Pclass groups is the best choice because of its high correlation with Age (0.408106)","metadata":{}},{"cell_type":"code","source":"age_by_pclass_sex = data_all.groupby(['Sex', 'Pclass']).median()['Age']\nage_by_pclass_sex","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for pclass in range(1 , 4):\n    for sex in ['female','male']:\n        print('Median age of Pclasse {} {}s : {}'.format(pclass , sex ,\n                                                         age_by_pclass_sex[sex][pclass] ))\n        \nprint('medain age of all passengers {}'.format(data_all['Age'].median()))\n\n# Filling the missing values in Age with the medians of Sex and Pclass groups\ndata_all['Age'] = data_all.groupby(['Sex','Pclass'])['Age'].apply(lambda x : x.fillna(x.median()))","metadata":{"execution":{"iopub.status.busy":"2022-07-16T05:27:01.444483Z","iopub.execute_input":"2022-07-16T05:27:01.444847Z","iopub.status.idle":"2022-07-16T05:27:01.464099Z","shell.execute_reply.started":"2022-07-16T05:27:01.444819Z","shell.execute_reply":"2022-07-16T05:27:01.463101Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#If you are having trouble understanding the code, just play with this code\nage_by_pclass_sex['female'][1] # female or male # 1, 2 or 3","metadata":{"execution":{"iopub.status.busy":"2022-07-16T05:27:14.895660Z","iopub.execute_input":"2022-07-16T05:27:14.896027Z","iopub.status.idle":"2022-07-16T05:27:14.903747Z","shell.execute_reply.started":"2022-07-16T05:27:14.895998Z","shell.execute_reply":"2022-07-16T05:27:14.902906Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_all.isnull().sum()['Age'] # just check","metadata":{"execution":{"iopub.status.busy":"2022-07-16T05:27:33.806005Z","iopub.execute_input":"2022-07-16T05:27:33.806324Z","iopub.status.idle":"2022-07-16T05:27:33.815797Z","shell.execute_reply.started":"2022-07-16T05:27:33.806301Z","shell.execute_reply":"2022-07-16T05:27:33.814399Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 1.2 Embarked\n\nEmbarked is a categorical feature and there are only 2 missing values in whole data set.oth of those passengers are female , upper class and they have the same ticket number -> value for an upper class female passenger is C (Cherbourg), but this doesn't necessarily mean that they embarked from that port","metadata":{}},{"cell_type":"code","source":"data_all[data_all['Embarked'].isnull()]","metadata":{"execution":{"iopub.status.busy":"2022-07-16T05:28:12.570597Z","iopub.execute_input":"2022-07-16T05:28:12.570974Z","iopub.status.idle":"2022-07-16T05:28:12.589834Z","shell.execute_reply.started":"2022-07-16T05:28:12.570945Z","shell.execute_reply":"2022-07-16T05:28:12.588642Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"When I googled Stone, Mrs. George Nelson (Martha Evelyn), I found that she embarked from S (Southampton) -> in this page https://www.encyclopedia-titanica.org/titanic-survivor/martha-evelyn-stone.html.","metadata":{}},{"cell_type":"code","source":"# Filling the missing values in Embarked with S\ndata_all['Embarked'] = data_all['Embarked'].fillna('s')","metadata":{"execution":{"iopub.status.busy":"2022-07-16T05:28:39.066857Z","iopub.execute_input":"2022-07-16T05:28:39.067166Z","iopub.status.idle":"2022-07-16T05:28:39.072774Z","shell.execute_reply.started":"2022-07-16T05:28:39.067144Z","shell.execute_reply":"2022-07-16T05:28:39.071541Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_all['Embarked'].isnull().sum() #check","metadata":{"execution":{"iopub.status.busy":"2022-07-16T05:28:49.458932Z","iopub.execute_input":"2022-07-16T05:28:49.459328Z","iopub.status.idle":"2022-07-16T05:28:49.468174Z","shell.execute_reply.started":"2022-07-16T05:28:49.459298Z","shell.execute_reply":"2022-07-16T05:28:49.467395Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 1.3 Fare\n1 missing ,We can assume that Fare is related to family size (Parch and SibSp and Pclass) features","metadata":{}},{"cell_type":"code","source":"data_all[data_all['Fare'].isnull()]","metadata":{"execution":{"iopub.status.busy":"2022-07-16T05:29:35.952130Z","iopub.execute_input":"2022-07-16T05:29:35.952437Z","iopub.status.idle":"2022-07-16T05:29:35.966796Z","shell.execute_reply.started":"2022-07-16T05:29:35.952414Z","shell.execute_reply":"2022-07-16T05:29:35.965762Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"med_fare = data_all.groupby(['Pclass', 'Parch', 'SibSp']).Fare.median()[3][0][0]\nmed_fare","metadata":{"execution":{"iopub.status.busy":"2022-07-16T05:29:53.408826Z","iopub.execute_input":"2022-07-16T05:29:53.409208Z","iopub.status.idle":"2022-07-16T05:29:53.420920Z","shell.execute_reply.started":"2022-07-16T05:29:53.409178Z","shell.execute_reply":"2022-07-16T05:29:53.419830Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"med_fare = data_all.groupby(['Pclass', 'Parch', 'SibSp']).Fare.median()[3][0][0]\n# Filling the missing value in Fare with the median Fare of 3rd class alone passenger\ndata_all['Fare'] = data_all['Fare'].fillna(med_fare)","metadata":{"execution":{"iopub.status.busy":"2022-07-16T05:30:02.203027Z","iopub.execute_input":"2022-07-16T05:30:02.203477Z","iopub.status.idle":"2022-07-16T05:30:02.213691Z","shell.execute_reply.started":"2022-07-16T05:30:02.203439Z","shell.execute_reply":"2022-07-16T05:30:02.212608Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 1.4 Cabin \nWe have more than 20% data loss , so drop it","metadata":{}},{"cell_type":"code","source":"# Dropping the Cabin feature\ndata_all.drop(['Cabin'], inplace=True, axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-07-16T05:30:39.031562Z","iopub.execute_input":"2022-07-16T05:30:39.031861Z","iopub.status.idle":"2022-07-16T05:30:39.037417Z","shell.execute_reply.started":"2022-07-16T05:30:39.031837Z","shell.execute_reply":"2022-07-16T05:30:39.036356Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(titanic_train.columns)\nprint(titanic_test.columns)","metadata":{"execution":{"iopub.status.busy":"2022-07-16T05:30:46.423365Z","iopub.execute_input":"2022-07-16T05:30:46.423743Z","iopub.status.idle":"2022-07-16T05:30:46.431089Z","shell.execute_reply.started":"2022-07-16T05:30:46.423714Z","shell.execute_reply":"2022-07-16T05:30:46.429609Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def divide_df(all_data):\n    # Returns divided dfs of training and test set\n    return all_data.loc[:890], all_data.loc[891:].drop(['Survived'], axis=1)\n\ntitanic_train, titanic_test = divide_df(data_all)","metadata":{"execution":{"iopub.status.busy":"2022-07-16T05:31:16.792790Z","iopub.execute_input":"2022-07-16T05:31:16.793271Z","iopub.status.idle":"2022-07-16T05:31:16.801035Z","shell.execute_reply.started":"2022-07-16T05:31:16.793241Z","shell.execute_reply":"2022-07-16T05:31:16.800282Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Training Set')\nprint(titanic_train.isnull().sum())\nprint(\"=\" *50)\nprint('Test Set')\nprint(titanic_test.isnull().sum())","metadata":{"execution":{"iopub.status.busy":"2022-07-16T05:31:33.323201Z","iopub.execute_input":"2022-07-16T05:31:33.323553Z","iopub.status.idle":"2022-07-16T05:31:33.339169Z","shell.execute_reply.started":"2022-07-16T05:31:33.323524Z","shell.execute_reply":"2022-07-16T05:31:33.337973Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Data Visualization","metadata":{}},{"cell_type":"code","source":"data_all.corr()","metadata":{"execution":{"iopub.status.busy":"2022-07-16T05:31:56.748528Z","iopub.execute_input":"2022-07-16T05:31:56.748905Z","iopub.status.idle":"2022-07-16T05:31:56.764879Z","shell.execute_reply.started":"2022-07-16T05:31:56.748875Z","shell.execute_reply":"2022-07-16T05:31:56.763762Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(20,10))\nsns.heatmap(titanic_train.corr(),annot=True ,cmap='coolwarm' , fmt = '.2f' )","metadata":{"execution":{"iopub.status.busy":"2022-07-16T05:33:00.897516Z","iopub.execute_input":"2022-07-16T05:33:00.897872Z","iopub.status.idle":"2022-07-16T05:33:01.354592Z","shell.execute_reply.started":"2022-07-16T05:33:00.897845Z","shell.execute_reply":"2022-07-16T05:33:01.353966Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(20,10))\nsns.heatmap(titanic_test.corr(),annot=True ,cmap='coolwarm' , fmt = '.2f')","metadata":{"execution":{"iopub.status.busy":"2022-07-16T05:33:35.263046Z","iopub.execute_input":"2022-07-16T05:33:35.263422Z","iopub.status.idle":"2022-07-16T05:33:35.652346Z","shell.execute_reply.started":"2022-07-16T05:33:35.263398Z","shell.execute_reply":"2022-07-16T05:33:35.651402Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cat_features = ['Embarked', 'Parch', 'Pclass', 'Sex', 'SibSp']\n\nfig, axs = plt.subplots(ncols=2, nrows=3, figsize=(20, 20))\nplt.subplots_adjust(right=1.5, top=1.25)\n\nfor i, feature in enumerate(cat_features, 1):    \n    plt.subplot(2, 3, i)\n    sns.countplot(x=feature, hue='Survived', data=titanic_train)\n    \n    plt.xlabel('{}'.format(feature), size=20, labelpad=15)\n    plt.ylabel('Passenger Count', size=20, labelpad=15)    \n    plt.tick_params(axis='x', labelsize=20)\n    plt.tick_params(axis='y', labelsize=20)\n    \n    plt.legend(['Not Survived', 'Survived'], loc='upper center', prop={'size': 18})\n    plt.title('Count of Survival in {} Feature'.format(feature), size=20, y=1.05)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-16T05:33:46.391023Z","iopub.execute_input":"2022-07-16T05:33:46.391360Z","iopub.status.idle":"2022-07-16T05:33:47.315100Z","shell.execute_reply.started":"2022-07-16T05:33:46.391335Z","shell.execute_reply":"2022-07-16T05:33:47.314424Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Visualizing survivals based on gender\ntitanic_train['Died'] = 1 - titanic_train['Survived']\ntitanic_train.groupby('Sex').agg('sum')[['Survived', 'Died']].plot(kind='bar',\n                                                           figsize=(10, 5),\n                                                           stacked=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-16T05:34:01.274687Z","iopub.execute_input":"2022-07-16T05:34:01.275028Z","iopub.status.idle":"2022-07-16T05:34:01.472417Z","shell.execute_reply.started":"2022-07-16T05:34:01.274999Z","shell.execute_reply":"2022-07-16T05:34:01.470775Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_all['Age'] = pd.qcut(data_all['Age'], 10)","metadata":{"execution":{"iopub.status.busy":"2022-07-16T05:34:16.429977Z","iopub.execute_input":"2022-07-16T05:34:16.430346Z","iopub.status.idle":"2022-07-16T05:34:16.444685Z","shell.execute_reply.started":"2022-07-16T05:34:16.430318Z","shell.execute_reply":"2022-07-16T05:34:16.443714Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, axs = plt.subplots(figsize=(22, 9))\nsns.countplot(x='Age', hue='Survived', data=data_all)\n\nplt.xlabel('Age', size=15, labelpad=20)\nplt.ylabel('Passenger Count', size=15, labelpad=20)\nplt.tick_params(axis='x', labelsize=15)\nplt.tick_params(axis='y', labelsize=15)\n\nplt.legend(['Not Survived', 'Survived'], loc='upper right', prop={'size': 15})\nplt.title('Survival Counts in {} Feature'.format('Age'), size=15, y=1.05)\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-16T05:34:23.648910Z","iopub.execute_input":"2022-07-16T05:34:23.649289Z","iopub.status.idle":"2022-07-16T05:34:23.906145Z","shell.execute_reply.started":"2022-07-16T05:34:23.649260Z","shell.execute_reply":"2022-07-16T05:34:23.905113Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Processing training data and apply ML model","metadata":{}},{"cell_type":"code","source":"#Cleaning the data by removing irrelevant columns\ndf1 =titanic_train.drop(['Name','Ticket','PassengerId','Died'], axis=1)\ndf1.head(5)","metadata":{"execution":{"iopub.status.busy":"2022-07-16T05:34:58.251590Z","iopub.execute_input":"2022-07-16T05:34:58.251906Z","iopub.status.idle":"2022-07-16T05:34:58.265468Z","shell.execute_reply.started":"2022-07-16T05:34:58.251882Z","shell.execute_reply":"2022-07-16T05:34:58.264630Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Converting the categorical features 'Sex' and 'Embarked' into numerical values 0 & 1\ndf1.Sex = df1.Sex.map({'female' : 0 ,\n                      'male': 1})\n\ndf1.Embarked = df1.Embarked.map({'S':0, 'C':1, 'Q':2 })\ndf1.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-16T05:36:00.186933Z","iopub.execute_input":"2022-07-16T05:36:00.187253Z","iopub.status.idle":"2022-07-16T05:36:00.203350Z","shell.execute_reply.started":"2022-07-16T05:36:00.187229Z","shell.execute_reply":"2022-07-16T05:36:00.202218Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df1.dropna(inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-16T05:36:11.333517Z","iopub.execute_input":"2022-07-16T05:36:11.333877Z","iopub.status.idle":"2022-07-16T05:36:11.343010Z","shell.execute_reply.started":"2022-07-16T05:36:11.333848Z","shell.execute_reply":"2022-07-16T05:36:11.341626Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Doing Feature Scaling to standardize the independent features present in the data in a fixed range\ndf1.Age = (df1.Age-min(df1.Age))/(max(df1.Age)-min(df1.Age))\ndf1.Fare = (df1.Fare-min(df1.Fare))/(max(df1.Fare)-min(df1.Fare))\ndf1.describe()","metadata":{"execution":{"iopub.status.busy":"2022-07-16T05:36:19.501312Z","iopub.execute_input":"2022-07-16T05:36:19.501723Z","iopub.status.idle":"2022-07-16T05:36:19.545207Z","shell.execute_reply.started":"2022-07-16T05:36:19.501692Z","shell.execute_reply":"2022-07-16T05:36:19.544033Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Creating models\n\n   - Logistic Regression\n   - KNN or k-Nearest Neighbors\n   - Support Vector Machines\n   - Naive Bayes classifier\n   - Decision Tree\n   - Random Forrest\n   - Perceptron\n   - Artificial neural network\n   - RVM or Relevance Vector Machine","metadata":{}},{"cell_type":"code","source":"#Splitting the data for training and testing\nfrom sklearn.model_selection import train_test_split\nX_train, X_test, y_train, y_test = train_test_split(\n    df1.drop(['Survived'], axis=1),\n    df1.Survived,\n    test_size= 0.2,\n    random_state=0,\n    stratify=df1.Survived)","metadata":{"execution":{"iopub.status.busy":"2022-07-16T05:36:54.057980Z","iopub.execute_input":"2022-07-16T05:36:54.059222Z","iopub.status.idle":"2022-07-16T05:36:54.068669Z","shell.execute_reply.started":"2022-07-16T05:36:54.059184Z","shell.execute_reply":"2022-07-16T05:36:54.067239Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Logistic Regression","metadata":{}},{"cell_type":"code","source":"from sklearn.linear_model import LogisticRegression\nlogreg  = LogisticRegression()\nlogreg.fit(X_train, y_train)\n\nfrom sklearn.metrics import accuracy_score\ny_predict = logreg = logreg.predict(X_test)\nscore_log = round(accuracy_score(y_test, y_predict)*100 ,2)\nscore_log ","metadata":{"execution":{"iopub.status.busy":"2022-07-16T05:37:17.747279Z","iopub.execute_input":"2022-07-16T05:37:17.747609Z","iopub.status.idle":"2022-07-16T05:37:17.775167Z","shell.execute_reply.started":"2022-07-16T05:37:17.747581Z","shell.execute_reply":"2022-07-16T05:37:17.774395Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Confusion Matrix\nfrom sklearn.metrics import confusion_matrix\ncma=confusion_matrix(y_test, y_predict)\nsns.heatmap(cma,annot=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-16T05:37:28.410747Z","iopub.execute_input":"2022-07-16T05:37:28.411116Z","iopub.status.idle":"2022-07-16T05:37:28.611096Z","shell.execute_reply.started":"2022-07-16T05:37:28.411028Z","shell.execute_reply":"2022-07-16T05:37:28.610128Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Support Vector Machines\nare supervised learning models with associated learning algorithms that analyze data used for classification and regression analysis","metadata":{}},{"cell_type":"code","source":"svc = SVC()\nsvc.fit(X_train,y_train)\nY_pred = svc.predict(X_test)\nscore_scv = round(svc.score(X_train, y_train)*100,2)\nscore_scv","metadata":{"execution":{"iopub.status.busy":"2022-07-16T05:37:54.982520Z","iopub.execute_input":"2022-07-16T05:37:54.982882Z","iopub.status.idle":"2022-07-16T05:37:55.075441Z","shell.execute_reply.started":"2022-07-16T05:37:54.982854Z","shell.execute_reply":"2022-07-16T05:37:55.074306Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### k-Nearest Neighbors algorithm\nA sample is classified by a majority vote of its neighbors, with the sample being assigned to the class most common among its k nearest neighbors (k is a positive integer, typically small). If k = 1, then the object is simply assigned to the class of that single nearest neighbor","metadata":{}},{"cell_type":"code","source":"knn = KNeighborsClassifier(n_neighbors = 3)\nknn.fit(X_train,y_train)\nY_pred = knn.predict(X_test)\nscore_knn = round(knn.score(X_train, y_train) * 100, 2)\nscore_knn","metadata":{"execution":{"iopub.status.busy":"2022-07-16T05:38:28.640261Z","iopub.execute_input":"2022-07-16T05:38:28.640619Z","iopub.status.idle":"2022-07-16T05:38:28.702339Z","shell.execute_reply.started":"2022-07-16T05:38:28.640591Z","shell.execute_reply":"2022-07-16T05:38:28.701221Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### naive Bayes\nclassifiers are a family of simple probabilistic classifiers based on applying Bayes' theorem with strong (naive) independence assumptions between the features","metadata":{}},{"cell_type":"code","source":"gaussian = GaussianNB()\ngaussian.fit(X_train, y_train)\nY_pred = gaussian.predict(X_test)\nscore_gaussian = round(gaussian.score(X_train, y_train) * 100, 2)\nscore_gaussian","metadata":{"execution":{"iopub.status.busy":"2022-07-16T05:39:17.325105Z","iopub.execute_input":"2022-07-16T05:39:17.325600Z","iopub.status.idle":"2022-07-16T05:39:17.345433Z","shell.execute_reply.started":"2022-07-16T05:39:17.325550Z","shell.execute_reply":"2022-07-16T05:39:17.342596Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Stochastic Gradient Descent","metadata":{}},{"cell_type":"code","source":"sgd = SGDClassifier()\nsgd.fit(X_train, y_train)\nY_pred = sgd.predict(X_test)\nscore_sgd = round(sgd.score(X_train, y_train) * 100, 2)\nscore_sgd","metadata":{"execution":{"iopub.status.busy":"2022-07-16T05:40:39.840591Z","iopub.execute_input":"2022-07-16T05:40:39.840949Z","iopub.status.idle":"2022-07-16T05:40:39.859624Z","shell.execute_reply.started":"2022-07-16T05:40:39.840920Z","shell.execute_reply":"2022-07-16T05:40:39.858519Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Decision Tree\nThis model uses a decision tree as a predictive model which maps features (tree branches) to conclusions about the target value (tree leaves)","metadata":{}},{"cell_type":"code","source":"decision_tree = DecisionTreeClassifier()\ndecision_tree.fit(X_train, y_train)\nY_pred = decision_tree.predict(X_test)\nscore_decision_tree = round(decision_tree.score(X_train, y_train) * 100, 2)\nscore_decision_tree","metadata":{"execution":{"iopub.status.busy":"2022-07-16T05:41:07.541817Z","iopub.execute_input":"2022-07-16T05:41:07.542209Z","iopub.status.idle":"2022-07-16T05:41:07.559430Z","shell.execute_reply.started":"2022-07-16T05:41:07.542179Z","shell.execute_reply":"2022-07-16T05:41:07.558429Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Random Forest\nRandom forests or random decision forests are an ensemble learning method for classification, regression and other tasks, that operate by constructing a multitude of decision trees (n_estimators=100) at training time and outputting the class that is the mode of the classes (classification) or mean prediction (regression) of the individual trees","metadata":{}},{"cell_type":"code","source":"random_forest = RandomForestClassifier(n_estimators=40)\nrandom_forest.fit(X_train, y_train)\nY_pred = random_forest.predict(X_test)\nscore_random_forest = round(random_forest.score(X_train, y_train) * 100, 2)\nscore_random_forest","metadata":{"execution":{"iopub.status.busy":"2022-07-16T05:41:31.626953Z","iopub.execute_input":"2022-07-16T05:41:31.627343Z","iopub.status.idle":"2022-07-16T05:41:31.713453Z","shell.execute_reply.started":"2022-07-16T05:41:31.627314Z","shell.execute_reply":"2022-07-16T05:41:31.712520Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Model evaluation\nWe can now rank our evaluation of all the models to choose the best one for our problem","metadata":{}},{"cell_type":"code","source":"models = pd.DataFrame({\n    'Model': ['Support Vector Machines', 'KNN', 'Logistic Regression', \n              'Random Forest', 'Naive Bayes',  \n              'Stochastic Gradient Decent',  \n              'Decision Tree'],\n    'Score': [score_scv, score_knn, score_log, \n              score_random_forest, score_gaussian,\n              score_sgd,  score_decision_tree]})\nmodels.sort_values(by='Score', ascending=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-16T05:41:57.685798Z","iopub.execute_input":"2022-07-16T05:41:57.686108Z","iopub.status.idle":"2022-07-16T05:41:57.698650Z","shell.execute_reply.started":"2022-07-16T05:41:57.686084Z","shell.execute_reply":"2022-07-16T05:41:57.697761Z"},"trusted":true},"execution_count":null,"outputs":[]}]}