{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Build a predictive model that answers the question: “what sorts of people were more likely to survive?” using passenger data (ie name, age, gender, socio-economic class, etc). ","metadata":{}},{"cell_type":"markdown","source":" **Using different classification algorithms to predict the survivability and comparing them for the best fit model**","metadata":{}},{"cell_type":"markdown","source":"First step is importing the necessary libraries and reading the data into a dataframe","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport seaborn as sns\nimport numpy as np\ndf = pd.read_csv(\"../input/titanic/train.csv\")\ndf.shape","metadata":{"execution":{"iopub.status.busy":"2022-08-04T17:35:47.732253Z","iopub.execute_input":"2022-08-04T17:35:47.732681Z","iopub.status.idle":"2022-08-04T17:35:47.750365Z","shell.execute_reply.started":"2022-08-04T17:35:47.732647Z","shell.execute_reply":"2022-08-04T17:35:47.748798Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-04T17:35:47.754445Z","iopub.execute_input":"2022-08-04T17:35:47.755360Z","iopub.status.idle":"2022-08-04T17:35:47.785091Z","shell.execute_reply.started":"2022-08-04T17:35:47.755308Z","shell.execute_reply":"2022-08-04T17:35:47.783469Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Checking for Null values","metadata":{}},{"cell_type":"code","source":"\nsns.heatmap(df.isnull(),cbar=False)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T17:35:47.787364Z","iopub.execute_input":"2022-08-04T17:35:47.787897Z","iopub.status.idle":"2022-08-04T17:35:48.161917Z","shell.execute_reply.started":"2022-08-04T17:35:47.787808Z","shell.execute_reply":"2022-08-04T17:35:48.160715Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2022-08-04T17:35:48.163715Z","iopub.execute_input":"2022-08-04T17:35:48.164081Z","iopub.status.idle":"2022-08-04T17:35:48.174845Z","shell.execute_reply.started":"2022-08-04T17:35:48.164034Z","shell.execute_reply":"2022-08-04T17:35:48.173493Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.corr()","metadata":{"execution":{"iopub.status.busy":"2022-08-04T17:35:48.179670Z","iopub.execute_input":"2022-08-04T17:35:48.180257Z","iopub.status.idle":"2022-08-04T17:35:48.204406Z","shell.execute_reply.started":"2022-08-04T17:35:48.180191Z","shell.execute_reply":"2022-08-04T17:35:48.203281Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.info() ","metadata":{"execution":{"iopub.status.busy":"2022-08-04T17:35:48.205986Z","iopub.execute_input":"2022-08-04T17:35:48.206781Z","iopub.status.idle":"2022-08-04T17:35:48.233713Z","shell.execute_reply.started":"2022-08-04T17:35:48.206730Z","shell.execute_reply":"2022-08-04T17:35:48.232548Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#dropping unwanted columns\ndf.drop(['PassengerId','Name','Ticket','Cabin','Embarked'],axis='columns',inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T17:35:48.235149Z","iopub.execute_input":"2022-08-04T17:35:48.236375Z","iopub.status.idle":"2022-08-04T17:35:48.252434Z","shell.execute_reply.started":"2022-08-04T17:35:48.236324Z","shell.execute_reply":"2022-08-04T17:35:48.250973Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-04T17:35:48.254512Z","iopub.execute_input":"2022-08-04T17:35:48.255194Z","iopub.status.idle":"2022-08-04T17:35:48.274256Z","shell.execute_reply.started":"2022-08-04T17:35:48.255152Z","shell.execute_reply":"2022-08-04T17:35:48.273330Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"##### Some Visualisations","metadata":{}},{"cell_type":"code","source":"import seaborn as sns\nsns.countplot(x=\"Survived\", data = df)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T17:35:48.275384Z","iopub.execute_input":"2022-08-04T17:35:48.276168Z","iopub.status.idle":"2022-08-04T17:35:48.432497Z","shell.execute_reply.started":"2022-08-04T17:35:48.276133Z","shell.execute_reply":"2022-08-04T17:35:48.431164Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.countplot(x=\"Survived\",hue = \"Sex\", data = df)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T17:35:48.434002Z","iopub.execute_input":"2022-08-04T17:35:48.435001Z","iopub.status.idle":"2022-08-04T17:35:48.615630Z","shell.execute_reply.started":"2022-08-04T17:35:48.434963Z","shell.execute_reply":"2022-08-04T17:35:48.614623Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Separating the data into dependent and independent variables","metadata":{}},{"cell_type":"code","source":"inputs = df.drop('Survived',axis='columns')\ntarget = df.Survived","metadata":{"execution":{"iopub.status.busy":"2022-08-04T17:35:48.826740Z","iopub.execute_input":"2022-08-04T17:35:48.827133Z","iopub.status.idle":"2022-08-04T17:35:48.833282Z","shell.execute_reply.started":"2022-08-04T17:35:48.827097Z","shell.execute_reply":"2022-08-04T17:35:48.832035Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Filling the null values of column age using mean of the existing age data","metadata":{}},{"cell_type":"code","source":"inputs.Age = inputs.Age.fillna(inputs.Age.mean())","metadata":{"execution":{"iopub.status.busy":"2022-08-04T17:35:48.834805Z","iopub.execute_input":"2022-08-04T17:35:48.835507Z","iopub.status.idle":"2022-08-04T17:35:48.848046Z","shell.execute_reply.started":"2022-08-04T17:35:48.835468Z","shell.execute_reply":"2022-08-04T17:35:48.847116Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"inputs.head()","metadata":{"scrolled":true,"execution":{"iopub.status.busy":"2022-08-04T17:35:48.852120Z","iopub.execute_input":"2022-08-04T17:35:48.853038Z","iopub.status.idle":"2022-08-04T17:35:48.869199Z","shell.execute_reply.started":"2022-08-04T17:35:48.852999Z","shell.execute_reply":"2022-08-04T17:35:48.868014Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Performing column transformation on columns \"PClass\" and \"Sex\" using OneHotEncoder","metadata":{}},{"cell_type":"code","source":"import numpy as np\nfrom sklearn.compose import ColumnTransformer\nfrom sklearn.preprocessing import OneHotEncoder\nct = ColumnTransformer(transformers=[('encoder', OneHotEncoder(), [0, 1])], remainder='passthrough')\ninputs= np.array(ct.fit_transform(inputs))","metadata":{"execution":{"iopub.status.busy":"2022-08-04T17:35:48.871318Z","iopub.execute_input":"2022-08-04T17:35:48.872253Z","iopub.status.idle":"2022-08-04T17:35:48.960656Z","shell.execute_reply.started":"2022-08-04T17:35:48.872200Z","shell.execute_reply":"2022-08-04T17:35:48.959512Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Performing train_test_split on the data","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split","metadata":{"execution":{"iopub.status.busy":"2022-08-04T17:35:48.961921Z","iopub.execute_input":"2022-08-04T17:35:48.962437Z","iopub.status.idle":"2022-08-04T17:35:49.028374Z","shell.execute_reply.started":"2022-08-04T17:35:48.962402Z","shell.execute_reply":"2022-08-04T17:35:49.026988Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train, X_test, y_train, y_test = train_test_split(inputs,target,test_size=0.2,random_state=1)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T17:35:49.029803Z","iopub.execute_input":"2022-08-04T17:35:49.030173Z","iopub.status.idle":"2022-08-04T17:35:49.037208Z","shell.execute_reply.started":"2022-08-04T17:35:49.030139Z","shell.execute_reply":"2022-08-04T17:35:49.036317Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(X_train)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T17:35:49.038442Z","iopub.execute_input":"2022-08-04T17:35:49.039552Z","iopub.status.idle":"2022-08-04T17:35:49.050452Z","shell.execute_reply.started":"2022-08-04T17:35:49.039502Z","shell.execute_reply":"2022-08-04T17:35:49.049155Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(X_test)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T17:35:49.051944Z","iopub.execute_input":"2022-08-04T17:35:49.052796Z","iopub.status.idle":"2022-08-04T17:35:49.062522Z","shell.execute_reply.started":"2022-08-04T17:35:49.052757Z","shell.execute_reply":"2022-08-04T17:35:49.061410Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(inputs)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T17:35:49.064061Z","iopub.execute_input":"2022-08-04T17:35:49.064702Z","iopub.status.idle":"2022-08-04T17:35:49.076317Z","shell.execute_reply.started":"2022-08-04T17:35:49.064665Z","shell.execute_reply":"2022-08-04T17:35:49.074991Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Logistic Regression","metadata":{}},{"cell_type":"code","source":"from sklearn.linear_model import LogisticRegression\nmodel=LogisticRegression()\nmodel.fit(X_train,y_train)\ny_pred=model.predict(X_test)\n\n\nScores = pd.DataFrame({'Actual':y_test,'Predictions':y_pred})\nprint(Scores)\n","metadata":{"execution":{"iopub.status.busy":"2022-08-04T17:48:55.675881Z","iopub.execute_input":"2022-08-04T17:48:55.676329Z","iopub.status.idle":"2022-08-04T17:48:55.720370Z","shell.execute_reply.started":"2022-08-04T17:48:55.676291Z","shell.execute_reply":"2022-08-04T17:48:55.719157Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import confusion_matrix, accuracy_score\ncm = confusion_matrix(y_test, y_pred)\nprint(cm)\naccuracy_score(y_test, y_pred)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T17:35:49.217926Z","iopub.execute_input":"2022-08-04T17:35:49.218298Z","iopub.status.idle":"2022-08-04T17:35:49.231917Z","shell.execute_reply.started":"2022-08-04T17:35:49.218263Z","shell.execute_reply":"2022-08-04T17:35:49.230646Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.heatmap(cm,annot=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T17:35:49.235512Z","iopub.execute_input":"2022-08-04T17:35:49.236451Z","iopub.status.idle":"2022-08-04T17:35:49.465562Z","shell.execute_reply.started":"2022-08-04T17:35:49.236404Z","shell.execute_reply":"2022-08-04T17:35:49.464368Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Decision Tree","metadata":{}},{"cell_type":"code","source":"from sklearn import tree\nmodel1 = tree.DecisionTreeClassifier()","metadata":{"execution":{"iopub.status.busy":"2022-08-04T17:35:49.467110Z","iopub.execute_input":"2022-08-04T17:35:49.467710Z","iopub.status.idle":"2022-08-04T17:35:49.554853Z","shell.execute_reply.started":"2022-08-04T17:35:49.467671Z","shell.execute_reply":"2022-08-04T17:35:49.553403Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model1.fit(X_train,y_train)\ny_pred1=model1.predict(X_test)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T17:35:49.556510Z","iopub.execute_input":"2022-08-04T17:35:49.556893Z","iopub.status.idle":"2022-08-04T17:35:49.568209Z","shell.execute_reply.started":"2022-08-04T17:35:49.556856Z","shell.execute_reply":"2022-08-04T17:35:49.567012Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cm1 = confusion_matrix(y_test, y_pred1)\nprint(cm1)\naccuracy_score(y_test, y_pred1)\n","metadata":{"execution":{"iopub.status.busy":"2022-08-04T17:35:49.569625Z","iopub.execute_input":"2022-08-04T17:35:49.570340Z","iopub.status.idle":"2022-08-04T17:35:49.581953Z","shell.execute_reply.started":"2022-08-04T17:35:49.570304Z","shell.execute_reply":"2022-08-04T17:35:49.580916Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.heatmap(cm1,annot=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T17:35:49.584271Z","iopub.execute_input":"2022-08-04T17:35:49.584785Z","iopub.status.idle":"2022-08-04T17:35:49.815035Z","shell.execute_reply.started":"2022-08-04T17:35:49.584738Z","shell.execute_reply":"2022-08-04T17:35:49.814148Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Support Vector Machine","metadata":{}},{"cell_type":"code","source":"from sklearn.svm import SVC\nmodel2 = SVC()\n\nmodel2.fit(X_train, y_train)\ny_pred2=model2.predict(X_test)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T17:35:49.816301Z","iopub.execute_input":"2022-08-04T17:35:49.817322Z","iopub.status.idle":"2022-08-04T17:35:49.849109Z","shell.execute_reply.started":"2022-08-04T17:35:49.817283Z","shell.execute_reply":"2022-08-04T17:35:49.848028Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cm2 = confusion_matrix(y_test, y_pred2)\nprint(cm2)\naccuracy_score(y_test, y_pred2)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T17:35:49.850534Z","iopub.execute_input":"2022-08-04T17:35:49.851161Z","iopub.status.idle":"2022-08-04T17:35:49.860390Z","shell.execute_reply.started":"2022-08-04T17:35:49.851121Z","shell.execute_reply":"2022-08-04T17:35:49.859359Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.heatmap(cm2,annot=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T17:35:49.861742Z","iopub.execute_input":"2022-08-04T17:35:49.862532Z","iopub.status.idle":"2022-08-04T17:35:50.094149Z","shell.execute_reply.started":"2022-08-04T17:35:49.862486Z","shell.execute_reply":"2022-08-04T17:35:50.092894Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Comparison of different models**\n\n1. Logistic regression\n   Accuracy=0.804\n\n2. Decision Tree\n   Accuracy=0.748\n   \n3. Support Vector machine\n   Accuracy=0.659","metadata":{}},{"cell_type":"markdown","source":"**Conclusion**\n\nAmong the three models used for prediction Logistic Regression has the highest accuracy.","metadata":{}}]}