{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## 1. Import the required libraries and modules:","metadata":{"papermill":{"duration":0.015478,"end_time":"2022-07-04T08:10:23.769152","exception":false,"start_time":"2022-07-04T08:10:23.753674","status":"completed"},"tags":[]}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\n%matplotlib inline\nimport seaborn as sns\n\n#Machine Learning\nfrom sklearn.preprocessing import StandardScaler, LabelEncoder\nfrom sklearn.model_selection import train_test_split, GridSearchCV, cross_val_score\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.tree import DecisionTreeClassifier\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.neighbors import KNeighborsClassifier\nfrom sklearn.naive_bayes import GaussianNB  \nfrom sklearn.metrics import confusion_matrix, accuracy_score\n\nimport warnings\nwarnings.filterwarnings('ignore')","metadata":{"papermill":{"duration":1.344031,"end_time":"2022-07-04T08:10:25.128823","exception":false,"start_time":"2022-07-04T08:10:23.784792","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-04T18:59:33.613866Z","iopub.execute_input":"2022-07-04T18:59:33.614299Z","iopub.status.idle":"2022-07-04T18:59:33.626337Z","shell.execute_reply.started":"2022-07-04T18:59:33.614261Z","shell.execute_reply":"2022-07-04T18:59:33.625189Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 2. Load the data:","metadata":{"papermill":{"duration":0.015621,"end_time":"2022-07-04T08:10:25.161079","exception":false,"start_time":"2022-07-04T08:10:25.145458","status":"completed"},"tags":[]}},{"cell_type":"code","source":"train_df = pd.read_csv('../input/titanic/train.csv')\ntest_df = pd.read_csv('../input/titanic/test.csv')","metadata":{"papermill":{"duration":0.044407,"end_time":"2022-07-04T08:10:25.221088","exception":false,"start_time":"2022-07-04T08:10:25.176681","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-04T18:59:33.628034Z","iopub.execute_input":"2022-07-04T18:59:33.628839Z","iopub.status.idle":"2022-07-04T18:59:33.665215Z","shell.execute_reply.started":"2022-07-04T18:59:33.628787Z","shell.execute_reply":"2022-07-04T18:59:33.663999Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"raw","source":"train_df.head()","metadata":{"papermill":{"duration":0.015517,"end_time":"2022-07-04T08:10:25.25219","exception":false,"start_time":"2022-07-04T08:10:25.236673","status":"completed"},"tags":[]}},{"cell_type":"code","source":"print(train_df.columns)","metadata":{"papermill":{"duration":0.022419,"end_time":"2022-07-04T08:10:25.345661","exception":false,"start_time":"2022-07-04T08:10:25.323242","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-04T18:59:33.667058Z","iopub.execute_input":"2022-07-04T18:59:33.667628Z","iopub.status.idle":"2022-07-04T18:59:33.673527Z","shell.execute_reply.started":"2022-07-04T18:59:33.667593Z","shell.execute_reply":"2022-07-04T18:59:33.672341Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.info()","metadata":{"papermill":{"duration":0.039562,"end_time":"2022-07-04T08:10:25.307237","exception":false,"start_time":"2022-07-04T08:10:25.267675","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-04T18:59:33.674965Z","iopub.execute_input":"2022-07-04T18:59:33.675535Z","iopub.status.idle":"2022-07-04T18:59:33.705026Z","shell.execute_reply.started":"2022-07-04T18:59:33.675502Z","shell.execute_reply":"2022-07-04T18:59:33.703688Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.describe(include=\"all\")","metadata":{"papermill":{"duration":0.068468,"end_time":"2022-07-04T08:10:25.429829","exception":false,"start_time":"2022-07-04T08:10:25.361361","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-04T18:59:33.708504Z","iopub.execute_input":"2022-07-04T18:59:33.709328Z","iopub.status.idle":"2022-07-04T18:59:33.766221Z","shell.execute_reply.started":"2022-07-04T18:59:33.709274Z","shell.execute_reply":"2022-07-04T18:59:33.764848Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(pd.isnull(train_df).sum() / len(train_df)*100)","metadata":{"papermill":{"duration":0.02865,"end_time":"2022-07-04T08:10:25.475026","exception":false,"start_time":"2022-07-04T08:10:25.446376","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-04T18:59:33.768783Z","iopub.execute_input":"2022-07-04T18:59:33.769582Z","iopub.status.idle":"2022-07-04T18:59:33.779114Z","shell.execute_reply.started":"2022-07-04T18:59:33.769533Z","shell.execute_reply":"2022-07-04T18:59:33.777897Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df['Age'].isnull().sum()","metadata":{"papermill":{"duration":0.036609,"end_time":"2022-07-04T08:10:25.539715","exception":false,"start_time":"2022-07-04T08:10:25.503106","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-04T18:59:33.780681Z","iopub.execute_input":"2022-07-04T18:59:33.78137Z","iopub.status.idle":"2022-07-04T18:59:33.798745Z","shell.execute_reply.started":"2022-07-04T18:59:33.781335Z","shell.execute_reply":"2022-07-04T18:59:33.79741Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Some observation, \n1. Numerical Features: Age (Continuous), Fare (Continuous), SibSp (Discrete), Parch (Discrete)\n2. Categorical Features: Survived, Sex, Embarked, Pclass\n3. Alphanumeric Features: Ticket, Cabin\n4. There are a total of 891 passengers in our training set.\n5. The Age feature is missing approximately 19.8% of its values. We should handle the missing values.\n6. The Cabin feature is missing approximately 77.1% of its values. We can drop this feature.\n7. The Embarked feature is missing 0.22% of its values, which should be relatively harmless so no action required.","metadata":{"papermill":{"duration":0.02687,"end_time":"2022-07-04T08:10:25.593812","exception":false,"start_time":"2022-07-04T08:10:25.566942","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"### Explore the target variable: Survived","metadata":{"papermill":{"duration":0.026992,"end_time":"2022-07-04T08:10:25.647874","exception":false,"start_time":"2022-07-04T08:10:25.620882","status":"completed"},"tags":[]}},{"cell_type":"code","source":"train_df.Survived.value_counts()","metadata":{"papermill":{"duration":0.035877,"end_time":"2022-07-04T08:10:25.711289","exception":false,"start_time":"2022-07-04T08:10:25.675412","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-04T18:59:33.801245Z","iopub.execute_input":"2022-07-04T18:59:33.801818Z","iopub.status.idle":"2022-07-04T18:59:33.815298Z","shell.execute_reply.started":"2022-07-04T18:59:33.801755Z","shell.execute_reply":"2022-07-04T18:59:33.814484Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.Survived.value_counts()/len(train_df) * 100","metadata":{"papermill":{"duration":0.037223,"end_time":"2022-07-04T08:10:25.775927","exception":false,"start_time":"2022-07-04T08:10:25.738704","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-04T18:59:33.816485Z","iopub.execute_input":"2022-07-04T18:59:33.816982Z","iopub.status.idle":"2022-07-04T18:59:33.833237Z","shell.execute_reply.started":"2022-07-04T18:59:33.816952Z","shell.execute_reply":"2022-07-04T18:59:33.832341Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Some observation,\n1. Since it has just two values 0 and 1 we won't be required to do any feature encoding to the target variable.\n2. Almost 62 % of people did not survived the titanic, so truely disastrous event.\n3. We will need to analysis this further with respect to different features.","metadata":{"papermill":{"duration":0.028693,"end_time":"2022-07-04T08:10:25.832275","exception":false,"start_time":"2022-07-04T08:10:25.803582","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"### Explore the independent variables:","metadata":{"papermill":{"duration":0.027624,"end_time":"2022-07-04T08:10:25.887753","exception":false,"start_time":"2022-07-04T08:10:25.860129","status":"completed"},"tags":[]}},{"cell_type":"code","source":"train_df.columns","metadata":{"papermill":{"duration":0.036038,"end_time":"2022-07-04T08:10:25.951682","exception":false,"start_time":"2022-07-04T08:10:25.915644","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-04T18:59:33.834576Z","iopub.execute_input":"2022-07-04T18:59:33.835095Z","iopub.status.idle":"2022-07-04T18:59:33.84813Z","shell.execute_reply.started":"2022-07-04T18:59:33.835064Z","shell.execute_reply":"2022-07-04T18:59:33.84706Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### 1) Passenger ID is the identity column and hence unique.It won't provide us any useful infromation so we will drop the column in data cleaning","metadata":{"papermill":{"duration":0.02896,"end_time":"2022-07-04T08:10:26.011225","exception":false,"start_time":"2022-07-04T08:10:25.982265","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"#### 2) Survived is our target variable.","metadata":{"papermill":{"duration":0.028542,"end_time":"2022-07-04T08:10:26.067628","exception":false,"start_time":"2022-07-04T08:10:26.039086","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"#### 3) Pclass","metadata":{"papermill":{"duration":0.027537,"end_time":"2022-07-04T08:10:26.123831","exception":false,"start_time":"2022-07-04T08:10:26.096294","status":"completed"},"tags":[]}},{"cell_type":"code","source":"train_df.Pclass.value_counts()","metadata":{"papermill":{"duration":0.039037,"end_time":"2022-07-04T08:10:26.191017","exception":false,"start_time":"2022-07-04T08:10:26.15198","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-04T18:59:33.849408Z","iopub.execute_input":"2022-07-04T18:59:33.849921Z","iopub.status.idle":"2022-07-04T18:59:33.867133Z","shell.execute_reply.started":"2022-07-04T18:59:33.84989Z","shell.execute_reply":"2022-07-04T18:59:33.865668Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.barplot(x=\"Pclass\", y=\"Survived\", data=train_df)\nPclass1 = train_df[\"Survived\"][train_df[\"Pclass\"] == 1].value_counts(normalize = True)[1]*100\nPclass2 = train_df[\"Survived\"][train_df[\"Pclass\"] == 2].value_counts(normalize = True)[1]*100\nPclass3 = train_df[\"Survived\"][train_df[\"Pclass\"] == 3].value_counts(normalize = True)[1]*100\nprint(f\"Percentage of Pclass 1 who survived: {Pclass1}\")\nprint(f\"Percentage of Pclass 2 who survived: {Pclass2}\")\nprint(f\"Percentage of Pclass 3 who survived: {Pclass3}\")","metadata":{"papermill":{"duration":0.283366,"end_time":"2022-07-04T08:10:26.49101","exception":false,"start_time":"2022-07-04T08:10:26.207644","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-04T18:59:33.868661Z","iopub.execute_input":"2022-07-04T18:59:33.869895Z","iopub.status.idle":"2022-07-04T18:59:34.210619Z","shell.execute_reply.started":"2022-07-04T18:59:33.869856Z","shell.execute_reply":"2022-07-04T18:59:34.209189Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Some observation\n1. Survival rate increases as the Pclass decreases or in other word, \n2. Upper class(1) has a higher chance of survival then the Middle(2) and Lower class(3) \n3. Lower class has lowest survival rate (Which is unfortunate but as expected)","metadata":{"papermill":{"duration":0.017014,"end_time":"2022-07-04T08:10:26.527407","exception":false,"start_time":"2022-07-04T08:10:26.510393","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"#### 4) Name this again a unique column and won't add any significant importance hence we can drop this as well.","metadata":{"papermill":{"duration":0.016857,"end_time":"2022-07-04T08:10:26.561338","exception":false,"start_time":"2022-07-04T08:10:26.544481","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"#### 5) Sex","metadata":{"papermill":{"duration":0.016716,"end_time":"2022-07-04T08:10:26.595044","exception":false,"start_time":"2022-07-04T08:10:26.578328","status":"completed"},"tags":[]}},{"cell_type":"code","source":"train_df.Sex.value_counts()","metadata":{"papermill":{"duration":0.025346,"end_time":"2022-07-04T08:10:26.638014","exception":false,"start_time":"2022-07-04T08:10:26.612668","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-04T18:59:34.211677Z","iopub.execute_input":"2022-07-04T18:59:34.212088Z","iopub.status.idle":"2022-07-04T18:59:34.221724Z","shell.execute_reply.started":"2022-07-04T18:59:34.211991Z","shell.execute_reply":"2022-07-04T18:59:34.220193Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.barplot(x=\"Sex\", y=\"Survived\", data=train_df)\n#print percentages of females vs. males that survive\nfemale = train_df[\"Survived\"][train_df[\"Sex\"] == 'female'].value_counts(normalize = True)[1]*100\nmale = train_df[\"Survived\"][train_df[\"Sex\"] == 'male'].value_counts(normalize = True)[1]*100\nprint(f\"Percentage of females who survived: {female}\")\nprint(f\"Percentage of males who survived: {male}\")","metadata":{"papermill":{"duration":0.176932,"end_time":"2022-07-04T08:10:26.831956","exception":false,"start_time":"2022-07-04T08:10:26.655024","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-04T18:59:34.223316Z","iopub.execute_input":"2022-07-04T18:59:34.223993Z","iopub.status.idle":"2022-07-04T18:59:34.482558Z","shell.execute_reply.started":"2022-07-04T18:59:34.223952Z","shell.execute_reply":"2022-07-04T18:59:34.481347Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Some observations\n1. Female have higher chance of survival then male which is expected as prefrence was given to child and womens.\n2. It will be intesting to see if pClass has any impact on female survival rate.","metadata":{"papermill":{"duration":0.019395,"end_time":"2022-07-04T08:10:26.874444","exception":false,"start_time":"2022-07-04T08:10:26.855049","status":"completed"},"tags":[]}},{"cell_type":"code","source":"sns.barplot(y = 'Survived', x = 'Sex', hue = 'Pclass', data=train_df)   ","metadata":{"papermill":{"duration":0.28478,"end_time":"2022-07-04T08:10:27.176723","exception":false,"start_time":"2022-07-04T08:10:26.891943","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-04T18:59:34.484294Z","iopub.execute_input":"2022-07-04T18:59:34.484638Z","iopub.status.idle":"2022-07-04T18:59:34.918341Z","shell.execute_reply.started":"2022-07-04T18:59:34.484604Z","shell.execute_reply":"2022-07-04T18:59:34.917152Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Some Observation\n1. Irrespective of class, Male always have lower chance of survival.\n2. Survival rate for Females does significantly lowered in lower class","metadata":{"papermill":{"duration":0.017401,"end_time":"2022-07-04T08:10:27.215366","exception":false,"start_time":"2022-07-04T08:10:27.197965","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"#### 6) Age","metadata":{"papermill":{"duration":0.017134,"end_time":"2022-07-04T08:10:27.249937","exception":false,"start_time":"2022-07-04T08:10:27.232803","status":"completed"},"tags":[]}},{"cell_type":"code","source":"train_df.Age.value_counts()","metadata":{"papermill":{"duration":0.03154,"end_time":"2022-07-04T08:10:27.298915","exception":false,"start_time":"2022-07-04T08:10:27.267375","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-04T18:59:34.926432Z","iopub.execute_input":"2022-07-04T18:59:34.92725Z","iopub.status.idle":"2022-07-04T18:59:34.947119Z","shell.execute_reply.started":"2022-07-04T18:59:34.927199Z","shell.execute_reply":"2022-07-04T18:59:34.945859Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Lets check top 5 age group\ntrain_df.Age.value_counts().nlargest(5).plot.barh(); ","metadata":{"papermill":{"duration":0.15022,"end_time":"2022-07-04T08:10:27.467063","exception":false,"start_time":"2022-07-04T08:10:27.316843","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-04T18:59:34.948972Z","iopub.execute_input":"2022-07-04T18:59:34.949874Z","iopub.status.idle":"2022-07-04T18:59:35.177081Z","shell.execute_reply.started":"2022-07-04T18:59:34.94982Z","shell.execute_reply":"2022-07-04T18:59:35.175845Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.histplot(x='Age', data=train_df, bins=20);","metadata":{"papermill":{"duration":0.194862,"end_time":"2022-07-04T08:10:27.686754","exception":false,"start_time":"2022-07-04T08:10:27.491892","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-04T18:59:35.178836Z","iopub.execute_input":"2022-07-04T18:59:35.179489Z","iopub.status.idle":"2022-07-04T18:59:35.458687Z","shell.execute_reply.started":"2022-07-04T18:59:35.179446Z","shell.execute_reply":"2022-07-04T18:59:35.45735Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### 7) Sibsp","metadata":{"papermill":{"duration":0.018101,"end_time":"2022-07-04T08:10:27.723615","exception":false,"start_time":"2022-07-04T08:10:27.705514","status":"completed"},"tags":[]}},{"cell_type":"code","source":"train_df.SibSp.value_counts()","metadata":{"papermill":{"duration":0.027292,"end_time":"2022-07-04T08:10:27.768992","exception":false,"start_time":"2022-07-04T08:10:27.7417","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-04T18:59:35.46043Z","iopub.execute_input":"2022-07-04T18:59:35.460896Z","iopub.status.idle":"2022-07-04T18:59:35.472072Z","shell.execute_reply.started":"2022-07-04T18:59:35.460853Z","shell.execute_reply":"2022-07-04T18:59:35.470645Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.barplot(x=\"SibSp\", y=\"Survived\", data=train_df)","metadata":{"papermill":{"duration":0.284274,"end_time":"2022-07-04T08:10:28.071908","exception":false,"start_time":"2022-07-04T08:10:27.787634","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-04T18:59:35.473863Z","iopub.execute_input":"2022-07-04T18:59:35.474585Z","iopub.status.idle":"2022-07-04T18:59:35.930268Z","shell.execute_reply.started":"2022-07-04T18:59:35.47454Z","shell.execute_reply":"2022-07-04T18:59:35.929143Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Some observation\n1. As number of sibling increases, chances of survival decreases.\n2. Exception to this is people having no sibling, they have survival rate lower than people with 1 and 2 siblings. \n3. May be this is because crew and staff members possibly have no sibling on board and are less likely to survive. But this is just a speculation.","metadata":{"papermill":{"duration":0.018027,"end_time":"2022-07-04T08:10:28.114309","exception":false,"start_time":"2022-07-04T08:10:28.096282","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"#### 8) Parch","metadata":{"papermill":{"duration":0.018059,"end_time":"2022-07-04T08:10:28.150828","exception":false,"start_time":"2022-07-04T08:10:28.132769","status":"completed"},"tags":[]}},{"cell_type":"code","source":"train_df.Parch.value_counts()","metadata":{"papermill":{"duration":0.027738,"end_time":"2022-07-04T08:10:28.197131","exception":false,"start_time":"2022-07-04T08:10:28.169393","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-04T18:59:35.931597Z","iopub.execute_input":"2022-07-04T18:59:35.931957Z","iopub.status.idle":"2022-07-04T18:59:35.940283Z","shell.execute_reply.started":"2022-07-04T18:59:35.931924Z","shell.execute_reply":"2022-07-04T18:59:35.939297Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.barplot(x=\"Parch\", y=\"Survived\", data=train_df)","metadata":{"papermill":{"duration":0.265097,"end_time":"2022-07-04T08:10:28.481156","exception":false,"start_time":"2022-07-04T08:10:28.216059","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-04T18:59:35.941312Z","iopub.execute_input":"2022-07-04T18:59:35.942111Z","iopub.status.idle":"2022-07-04T18:59:36.341167Z","shell.execute_reply.started":"2022-07-04T18:59:35.942061Z","shell.execute_reply":"2022-07-04T18:59:36.340148Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Some Observations:\n1. Pepole travelling with more than 4 parent or child are less likely to survive.\n2. Also people who don't have any parent or child are also survived less than people having 1 to 3 parent/child. This is consistant with Sibling data","metadata":{"papermill":{"duration":0.018707,"end_time":"2022-07-04T08:10:28.524477","exception":false,"start_time":"2022-07-04T08:10:28.50577","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"#### 9) Ticket, unique column less likley to impact survival so dropping the feature.","metadata":{"papermill":{"duration":0.019739,"end_time":"2022-07-04T08:10:28.563177","exception":false,"start_time":"2022-07-04T08:10:28.543438","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"#### 10) Fare\n#### fare is discreate numerical feature we may need to do feature engineering for optimize result","metadata":{"papermill":{"duration":0.018998,"end_time":"2022-07-04T08:10:28.601316","exception":false,"start_time":"2022-07-04T08:10:28.582318","status":"completed"},"tags":[]}},{"cell_type":"code","source":"sns.distplot(train_df['Fare'], bins = 20, kde = True, vertical = False) ","metadata":{"papermill":{"duration":0.201652,"end_time":"2022-07-04T08:10:28.822809","exception":false,"start_time":"2022-07-04T08:10:28.621157","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-04T18:59:36.342663Z","iopub.execute_input":"2022-07-04T18:59:36.343029Z","iopub.status.idle":"2022-07-04T18:59:36.59165Z","shell.execute_reply.started":"2022-07-04T18:59:36.342998Z","shell.execute_reply":"2022-07-04T18:59:36.59015Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### 11) Cabin, Cabin has high percentage of missing data, we will drop this feature","metadata":{"papermill":{"duration":0.026367,"end_time":"2022-07-04T08:10:28.86918","exception":false,"start_time":"2022-07-04T08:10:28.842813","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"#### 12) Embarked","metadata":{"papermill":{"duration":0.018607,"end_time":"2022-07-04T08:10:28.906819","exception":false,"start_time":"2022-07-04T08:10:28.888212","status":"completed"},"tags":[]}},{"cell_type":"code","source":"train_df.Embarked.value_counts()","metadata":{"papermill":{"duration":0.02874,"end_time":"2022-07-04T08:10:28.954395","exception":false,"start_time":"2022-07-04T08:10:28.925655","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-04T18:59:36.593449Z","iopub.execute_input":"2022-07-04T18:59:36.593799Z","iopub.status.idle":"2022-07-04T18:59:36.602582Z","shell.execute_reply.started":"2022-07-04T18:59:36.593747Z","shell.execute_reply":"2022-07-04T18:59:36.601183Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.barplot(x=\"Embarked\", y=\"Survived\", data=train_df)\nembS = train_df[\"Survived\"][train_df[\"Embarked\"] == 'S'].value_counts(normalize = True)[1]*100\nembC = train_df[\"Survived\"][train_df[\"Embarked\"] == 'C'].value_counts(normalize = True)[1]*100\nembQ = train_df[\"Survived\"][train_df[\"Embarked\"] == 'Q'].value_counts(normalize = True)[1]*100\nprint(f\"Percentage of Embarked S who survived: {embS}\")\nprint(f\"Percentage of Embarked C who survived: {embC}\")\nprint(f\"Percentage of Embarked Q who survived: {embQ}\")","metadata":{"papermill":{"duration":0.191721,"end_time":"2022-07-04T08:10:29.165435","exception":false,"start_time":"2022-07-04T08:10:28.973714","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-04T18:59:36.604136Z","iopub.execute_input":"2022-07-04T18:59:36.604491Z","iopub.status.idle":"2022-07-04T18:59:36.892245Z","shell.execute_reply.started":"2022-07-04T18:59:36.604459Z","shell.execute_reply":"2022-07-04T18:59:36.891019Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Some Observation\n1. People have embarked mostly from Southampton, but they have lowest survival rate\n2. People who embarked from Cherbourg have most survival rate.","metadata":{"papermill":{"duration":0.021933,"end_time":"2022-07-04T08:10:29.213163","exception":false,"start_time":"2022-07-04T08:10:29.19123","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"# Conclusion of EDA\n1. Age is important feture but has missing values, we will need to handle it while data preprocessing.\n2. PassengerId is Identity column, we will drop it.\n3. Survived is the target variable\n4. Name is Unique column, we will drop it.\n5. We will need to perform feture encoding on Sex and Embarked. We can use getDummies funtion from pandas library\n6. Females have high rate of surviver than Males\n7. As number of Sibling or parent/child increase, survival rate decrease with exception of people having no sibling or travelling alone.\n8. Ticket is Unique column, we will drop it.\n9. Cabin has high percentage of missing data, hence we will drop it.","metadata":{"papermill":{"duration":0.019045,"end_time":"2022-07-04T08:10:29.251606","exception":false,"start_time":"2022-07-04T08:10:29.232561","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"# 3. Data Preprocessing","metadata":{"papermill":{"duration":0.018723,"end_time":"2022-07-04T08:10:29.289503","exception":false,"start_time":"2022-07-04T08:10:29.27078","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"### Handle missing values","metadata":{"papermill":{"duration":0.019233,"end_time":"2022-07-04T08:10:29.327772","exception":false,"start_time":"2022-07-04T08:10:29.308539","status":"completed"},"tags":[]}},{"cell_type":"code","source":"print(train_df['Embarked'].isnull().sum())","metadata":{"papermill":{"duration":0.027409,"end_time":"2022-07-04T08:10:29.374161","exception":false,"start_time":"2022-07-04T08:10:29.346752","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-04T18:59:36.893896Z","iopub.execute_input":"2022-07-04T18:59:36.894242Z","iopub.status.idle":"2022-07-04T18:59:36.900811Z","shell.execute_reply.started":"2022-07-04T18:59:36.894212Z","shell.execute_reply":"2022-07-04T18:59:36.899532Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# There are only 2 missing values so we can fill them with mode value\n# replacing the missing values in the Embarked feature with S. as S is the largest\ntrain_df = train_df.fillna({\"Embarked\": \"S\"})","metadata":{"papermill":{"duration":0.025892,"end_time":"2022-07-04T08:10:29.419909","exception":false,"start_time":"2022-07-04T08:10:29.394017","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-04T18:59:36.902735Z","iopub.execute_input":"2022-07-04T18:59:36.903648Z","iopub.status.idle":"2022-07-04T18:59:36.918099Z","shell.execute_reply.started":"2022-07-04T18:59:36.903601Z","shell.execute_reply":"2022-07-04T18:59:36.916433Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(train_df['Embarked'].isnull().sum())","metadata":{"papermill":{"duration":0.026791,"end_time":"2022-07-04T08:10:29.465953","exception":false,"start_time":"2022-07-04T08:10:29.439162","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-04T18:59:36.919806Z","iopub.execute_input":"2022-07-04T18:59:36.92095Z","iopub.status.idle":"2022-07-04T18:59:36.930511Z","shell.execute_reply.started":"2022-07-04T18:59:36.92091Z","shell.execute_reply":"2022-07-04T18:59:36.929528Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df['Age'].isnull().sum()","metadata":{"papermill":{"duration":0.027232,"end_time":"2022-07-04T08:10:29.512426","exception":false,"start_time":"2022-07-04T08:10:29.485194","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-04T18:59:36.931693Z","iopub.execute_input":"2022-07-04T18:59:36.932371Z","iopub.status.idle":"2022-07-04T18:59:36.946814Z","shell.execute_reply.started":"2022-07-04T18:59:36.932329Z","shell.execute_reply":"2022-07-04T18:59:36.945503Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Since no of missing values are more we will use mean to fill the missing vlaues\ntrain_df.Age = train_df.Age.fillna(value=train_df.Age.mean())","metadata":{"papermill":{"duration":0.027102,"end_time":"2022-07-04T08:10:29.560528","exception":false,"start_time":"2022-07-04T08:10:29.533426","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-04T18:59:36.948759Z","iopub.execute_input":"2022-07-04T18:59:36.949622Z","iopub.status.idle":"2022-07-04T18:59:36.958103Z","shell.execute_reply.started":"2022-07-04T18:59:36.949585Z","shell.execute_reply":"2022-07-04T18:59:36.95671Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df['Age'].isnull().sum()","metadata":{"papermill":{"duration":0.028469,"end_time":"2022-07-04T08:10:29.608764","exception":false,"start_time":"2022-07-04T08:10:29.580295","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-04T18:59:36.959691Z","iopub.execute_input":"2022-07-04T18:59:36.960332Z","iopub.status.idle":"2022-07-04T18:59:36.971303Z","shell.execute_reply.started":"2022-07-04T18:59:36.960299Z","shell.execute_reply":"2022-07-04T18:59:36.97033Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Data Cleaning","metadata":{"papermill":{"duration":0.019279,"end_time":"2022-07-04T08:10:29.648461","exception":false,"start_time":"2022-07-04T08:10:29.629182","status":"completed"},"tags":[]}},{"cell_type":"code","source":"# Drop the unrequired columns - 'Name', 'Ticket' and 'Cabin'\n\ntrain_df = train_df.drop(['Name'], axis = 1)\ntest_df = test_df.drop(['Name'], axis = 1)\n\ntrain_df = train_df.drop(['Ticket'], axis = 1)\ntest_df = test_df.drop(['Ticket'], axis = 1)\n\ntrain_df = train_df.drop(['Cabin'], axis = 1)\ntest_df = test_df.drop(['Cabin'], axis = 1)","metadata":{"papermill":{"duration":0.032499,"end_time":"2022-07-04T08:10:29.700289","exception":false,"start_time":"2022-07-04T08:10:29.66779","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-04T18:59:36.97326Z","iopub.execute_input":"2022-07-04T18:59:36.974244Z","iopub.status.idle":"2022-07-04T18:59:36.988359Z","shell.execute_reply.started":"2022-07-04T18:59:36.974149Z","shell.execute_reply":"2022-07-04T18:59:36.98723Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.head()","metadata":{"papermill":{"duration":0.033564,"end_time":"2022-07-04T08:10:29.753951","exception":false,"start_time":"2022-07-04T08:10:29.720387","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-04T18:59:36.990111Z","iopub.execute_input":"2022-07-04T18:59:36.991359Z","iopub.status.idle":"2022-07-04T18:59:37.014371Z","shell.execute_reply.started":"2022-07-04T18:59:36.991311Z","shell.execute_reply":"2022-07-04T18:59:37.013018Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df.head()","metadata":{"papermill":{"duration":0.033422,"end_time":"2022-07-04T08:10:29.807395","exception":false,"start_time":"2022-07-04T08:10:29.773973","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-04T18:59:37.016092Z","iopub.execute_input":"2022-07-04T18:59:37.017277Z","iopub.status.idle":"2022-07-04T18:59:37.039954Z","shell.execute_reply.started":"2022-07-04T18:59:37.017224Z","shell.execute_reply":"2022-07-04T18:59:37.039055Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Feature Encoding:","metadata":{"papermill":{"duration":0.019502,"end_time":"2022-07-04T08:10:29.846443","exception":false,"start_time":"2022-07-04T08:10:29.826941","status":"completed"},"tags":[]}},{"cell_type":"code","source":"train_df.head()","metadata":{"papermill":{"duration":0.033769,"end_time":"2022-07-04T08:10:29.899757","exception":false,"start_time":"2022-07-04T08:10:29.865988","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-04T18:59:37.041275Z","iopub.execute_input":"2022-07-04T18:59:37.042081Z","iopub.status.idle":"2022-07-04T18:59:37.062556Z","shell.execute_reply.started":"2022-07-04T18:59:37.042026Z","shell.execute_reply":"2022-07-04T18:59:37.06124Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df.head()","metadata":{"papermill":{"duration":0.032206,"end_time":"2022-07-04T08:10:29.951689","exception":false,"start_time":"2022-07-04T08:10:29.919483","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-04T18:59:37.064119Z","iopub.execute_input":"2022-07-04T18:59:37.064846Z","iopub.status.idle":"2022-07-04T18:59:37.084741Z","shell.execute_reply.started":"2022-07-04T18:59:37.064791Z","shell.execute_reply":"2022-07-04T18:59:37.083549Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Here we are doing onehotencoding for categorical columns such as Sex, Embarked\ntrain_df = pd.get_dummies(train_df,drop_first=True)\ntest_df = pd.get_dummies(test_df,drop_first=True)","metadata":{"papermill":{"duration":0.034101,"end_time":"2022-07-04T08:10:30.005996","exception":false,"start_time":"2022-07-04T08:10:29.971895","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-04T18:59:37.086696Z","iopub.execute_input":"2022-07-04T18:59:37.087075Z","iopub.status.idle":"2022-07-04T18:59:37.10799Z","shell.execute_reply.started":"2022-07-04T18:59:37.087041Z","shell.execute_reply":"2022-07-04T18:59:37.106474Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.head()","metadata":{"papermill":{"duration":0.033909,"end_time":"2022-07-04T08:10:30.060427","exception":false,"start_time":"2022-07-04T08:10:30.026518","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-04T18:59:37.109875Z","iopub.execute_input":"2022-07-04T18:59:37.110622Z","iopub.status.idle":"2022-07-04T18:59:37.127236Z","shell.execute_reply.started":"2022-07-04T18:59:37.110573Z","shell.execute_reply":"2022-07-04T18:59:37.125866Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df.head()","metadata":{"papermill":{"duration":0.03535,"end_time":"2022-07-04T08:10:30.116542","exception":false,"start_time":"2022-07-04T08:10:30.081192","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-04T18:59:37.135441Z","iopub.execute_input":"2022-07-04T18:59:37.136049Z","iopub.status.idle":"2022-07-04T18:59:37.151414Z","shell.execute_reply.started":"2022-07-04T18:59:37.136004Z","shell.execute_reply":"2022-07-04T18:59:37.150253Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Feature Scaling:","metadata":{"papermill":{"duration":0.020322,"end_time":"2022-07-04T08:10:30.157985","exception":false,"start_time":"2022-07-04T08:10:30.137663","status":"completed"},"tags":[]}},{"cell_type":"code","source":"## independent and dependent features\n\n\nX = train_df.drop(['Survived', 'PassengerId'], axis=1)\ny = train_df[\"Survived\"]","metadata":{"papermill":{"duration":0.030182,"end_time":"2022-07-04T08:10:30.208659","exception":false,"start_time":"2022-07-04T08:10:30.178477","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-04T18:59:37.152628Z","iopub.execute_input":"2022-07-04T18:59:37.153711Z","iopub.status.idle":"2022-07-04T18:59:37.164653Z","shell.execute_reply.started":"2022-07-04T18:59:37.15366Z","shell.execute_reply":"2022-07-04T18:59:37.16365Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"### Applying Standard scaling to get optimized result\nsc = StandardScaler()\nX = sc.fit_transform(X)","metadata":{"papermill":{"duration":0.033164,"end_time":"2022-07-04T08:10:30.262804","exception":false,"start_time":"2022-07-04T08:10:30.22964","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-04T18:59:37.166463Z","iopub.execute_input":"2022-07-04T18:59:37.166874Z","iopub.status.idle":"2022-07-04T18:59:37.179591Z","shell.execute_reply.started":"2022-07-04T18:59:37.166841Z","shell.execute_reply":"2022-07-04T18:59:37.177976Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#  4. Train Test Data Split\n- Here the dataset will first need to be separated into X = independent variables and y = target variable.\n- Then dataset will be split into train and test datasets. We will use 80 - 20 split\n- random_state = 42 is used for reproduction\n- Fun Fact: Number 42 is used as an inside joke in the scientific and sci-fi community and it is derived from 'Hitchhiker’s Guide to the Galaxy. The number 42 also has a Wikipedia page for many pop culture references.","metadata":{"papermill":{"duration":0.020647,"end_time":"2022-07-04T08:10:30.304647","exception":false,"start_time":"2022-07-04T08:10:30.284","status":"completed"},"tags":[]}},{"cell_type":"code","source":"X_train, X_test, y_train, y_test = train_test_split(X, y, test_size = 0.2, random_state = 42)","metadata":{"papermill":{"duration":0.030406,"end_time":"2022-07-04T08:10:30.355614","exception":false,"start_time":"2022-07-04T08:10:30.325208","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-04T18:59:37.181184Z","iopub.execute_input":"2022-07-04T18:59:37.182306Z","iopub.status.idle":"2022-07-04T18:59:37.190045Z","shell.execute_reply.started":"2022-07-04T18:59:37.182111Z","shell.execute_reply":"2022-07-04T18:59:37.188816Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 5. Model Selection, Model Training and prediction","metadata":{"papermill":{"duration":0.020269,"end_time":"2022-07-04T08:10:30.396542","exception":false,"start_time":"2022-07-04T08:10:30.376273","status":"completed"},"tags":[]}},{"cell_type":"code","source":"\n# First initialize the list variables to store different outputs\n# We will use this later to compare all the different ML alorithms \nlst_model = []\nlst_accuracy  = []\nlst_accuracy_train = []\nlst_accuracy_test = [] \nlst_cv_score = []\nlst_TP = []\nlst_TN = []\nlst_FP = []\nlst_FN = []\n\n# fuction : accepts input model which is nothing but object instantiated of an algos\ndef applyMLmodel(model):\n    # train the model\n    model.fit(X_train, y_train)\n    accuracy = model.score(X_test, y_test) * 100\n    lst_accuracy.append(accuracy)\n    print(\"Accuracy :\", accuracy)\n    \n    # cross-validation , y_train.ravel() is similar to y_train.reshape(-1)\n    cv = cross_val_score(estimator = model, X = X_train, y = y_train.ravel(), cv = 10)\n    lst_cv_score.append(cv.mean())\n    print(\"CV Score :\", cv.mean())\n    \n    # predicting accuracy for training data set\n    y_pred_train = model.predict(X_train)\n    accuracy_train = accuracy_score(y_train, y_pred_train)\n    lst_accuracy_train.append(accuracy_train)\n    print(\"Accuracy(Training) :\", accuracy_train)\n\n    # predicting accuracy for test data set\n    y_pred_test = model.predict(X_test)\n    accuracy_test = accuracy_score(y_test, y_pred_test)\n    lst_accuracy_test.append(accuracy_test)\n    print(\"Accuracy(Test) :\", accuracy_test)\n\n    # confusion matrix\n    cm = confusion_matrix(y_test, y_pred_test)\n    print(\"Confusion Matrix :\")\n    print(cm)\n\n    # storing TN,TP,FN and FP as a part of list\n    lst_TN.append(cm[0,0])\n    lst_FP.append(cm[0,1])\n    lst_FN.append(cm[1,0])\n    lst_TP.append(cm[1,1])","metadata":{"papermill":{"duration":0.034061,"end_time":"2022-07-04T08:10:30.45232","exception":false,"start_time":"2022-07-04T08:10:30.418259","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-04T18:59:37.19186Z","iopub.execute_input":"2022-07-04T18:59:37.192561Z","iopub.status.idle":"2022-07-04T18:59:37.206977Z","shell.execute_reply.started":"2022-07-04T18:59:37.192517Z","shell.execute_reply":"2022-07-04T18:59:37.205833Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Now applying the different Models","metadata":{"papermill":{"duration":0.020483,"end_time":"2022-07-04T08:10:30.493624","exception":false,"start_time":"2022-07-04T08:10:30.473141","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"#### LogisticRegression","metadata":{"papermill":{"duration":0.020983,"end_time":"2022-07-04T08:10:30.535498","exception":false,"start_time":"2022-07-04T08:10:30.514515","status":"completed"},"tags":[]}},{"cell_type":"code","source":"model = LogisticRegression()\napplyMLmodel(model) \nlst_model.append(\"LogisticRegression\")","metadata":{"papermill":{"duration":0.078445,"end_time":"2022-07-04T08:10:30.634989","exception":false,"start_time":"2022-07-04T08:10:30.556544","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-04T18:59:37.208558Z","iopub.execute_input":"2022-07-04T18:59:37.20905Z","iopub.status.idle":"2022-07-04T18:59:37.290301Z","shell.execute_reply.started":"2022-07-04T18:59:37.209005Z","shell.execute_reply":"2022-07-04T18:59:37.289484Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### DecisionTreeClassifier","metadata":{"papermill":{"duration":0.034004,"end_time":"2022-07-04T08:10:30.70436","exception":false,"start_time":"2022-07-04T08:10:30.670356","status":"completed"},"tags":[]}},{"cell_type":"code","source":"model = DecisionTreeClassifier()\napplyMLmodel(model)\nlst_model.append(\"DecisionTreeClassifier\")","metadata":{"papermill":{"duration":0.074998,"end_time":"2022-07-04T08:10:30.813541","exception":false,"start_time":"2022-07-04T08:10:30.738543","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-04T18:59:37.292064Z","iopub.execute_input":"2022-07-04T18:59:37.292527Z","iopub.status.idle":"2022-07-04T18:59:37.333378Z","shell.execute_reply.started":"2022-07-04T18:59:37.292482Z","shell.execute_reply":"2022-07-04T18:59:37.332456Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### RandomForestClassifier","metadata":{"papermill":{"duration":0.034304,"end_time":"2022-07-04T08:10:30.883225","exception":false,"start_time":"2022-07-04T08:10:30.848921","status":"completed"},"tags":[]}},{"cell_type":"code","source":"modelR = RandomForestClassifier()\napplyMLmodel(modelR)\nlst_model.append(\"RandomForestClassifier\")","metadata":{"papermill":{"duration":1.709299,"end_time":"2022-07-04T08:10:32.626766","exception":false,"start_time":"2022-07-04T08:10:30.917467","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-04T18:59:37.334739Z","iopub.execute_input":"2022-07-04T18:59:37.335318Z","iopub.status.idle":"2022-07-04T18:59:39.825604Z","shell.execute_reply.started":"2022-07-04T18:59:37.335285Z","shell.execute_reply":"2022-07-04T18:59:39.824391Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### KNeighborsClassifier","metadata":{"papermill":{"duration":0.021005,"end_time":"2022-07-04T08:10:32.669182","exception":false,"start_time":"2022-07-04T08:10:32.648177","status":"completed"},"tags":[]}},{"cell_type":"code","source":"model =KNeighborsClassifier()  \napplyMLmodel(model)\nlst_model.append(\"KNeighborsClassifier\")","metadata":{"papermill":{"duration":0.094138,"end_time":"2022-07-04T08:10:32.784459","exception":false,"start_time":"2022-07-04T08:10:32.690321","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-04T18:59:39.827295Z","iopub.execute_input":"2022-07-04T18:59:39.827748Z","iopub.status.idle":"2022-07-04T18:59:39.937031Z","shell.execute_reply.started":"2022-07-04T18:59:39.827703Z","shell.execute_reply":"2022-07-04T18:59:39.935867Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### GaussianNB","metadata":{"papermill":{"duration":0.02079,"end_time":"2022-07-04T08:10:32.82705","exception":false,"start_time":"2022-07-04T08:10:32.80626","status":"completed"},"tags":[]}},{"cell_type":"code","source":"model = GaussianNB()\napplyMLmodel(model)\nlst_model.append(\"GaussianNB\")","metadata":{"papermill":{"duration":0.046138,"end_time":"2022-07-04T08:10:32.894119","exception":false,"start_time":"2022-07-04T08:10:32.847981","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-04T18:59:39.938262Z","iopub.execute_input":"2022-07-04T18:59:39.938583Z","iopub.status.idle":"2022-07-04T18:59:39.969252Z","shell.execute_reply.started":"2022-07-04T18:59:39.938554Z","shell.execute_reply":"2022-07-04T18:59:39.967987Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"### Comparing and visaulizing the results of ML algorithms","metadata":{"papermill":{"duration":0.028187,"end_time":"2022-07-04T08:10:32.943562","exception":false,"start_time":"2022-07-04T08:10:32.915375","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-04T18:59:39.971545Z","iopub.execute_input":"2022-07-04T18:59:39.972299Z","iopub.status.idle":"2022-07-04T18:59:39.977156Z","shell.execute_reply.started":"2022-07-04T18:59:39.972257Z","shell.execute_reply":"2022-07-04T18:59:39.976142Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# We will first convert all the list of outputs into dataframe which will be easy to compare and visuallize\npredictiondf = pd.DataFrame({'Model': np.array(lst_model),\n                             'Accuracy': np.array(lst_accuracy),\n                             'Accuracy(Training)' : np.array(lst_accuracy_train),\n                             'Accuracy(Test)' : np.array(lst_accuracy_test),\n                             'CV Score' : np.array(lst_cv_score),\n                             'True Positive' : np.array(lst_TP),\n                             'True Negative' : np.array(lst_TN),\n                             'False Positive' : np.array(lst_FP),\n                             'False Negative' : np.array(lst_FN)\n                            })\npredictiondf","metadata":{"papermill":{"duration":0.038414,"end_time":"2022-07-04T08:10:33.003291","exception":false,"start_time":"2022-07-04T08:10:32.964877","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-04T18:59:39.978439Z","iopub.execute_input":"2022-07-04T18:59:39.97875Z","iopub.status.idle":"2022-07-04T18:59:40.006204Z","shell.execute_reply.started":"2022-07-04T18:59:39.978714Z","shell.execute_reply":"2022-07-04T18:59:40.004819Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots(2,2, figsize=(24,14))\nplt.subplots_adjust(wspace = .25, hspace = .25)\n#comparing CV score\npredictiondf.sort_values(by=['CV Score'], ascending=False, inplace=True)\n\nsns.barplot(x='CV Score', y='Model', data = predictiondf, ax = ax[0][0])\nax[0][0].set_xlabel('Cross-Validaton Score', size=16)\nax[0][0].set_ylabel('Model')\nax[0][0].set_xlim(0,1.0)\nax[0][0].set_xticks(np.arange(0, 1.1, 0.1))\n\n#comparing accuracy\npredictiondf.sort_values(by=['Accuracy'], ascending=False, inplace=True)\n\nsns.barplot(x='Accuracy', y='Model', data = predictiondf, ax = ax[0][1])\nax[0][1].set_xlabel('Accuracy', size=16)\nax[0][1].set_ylabel('Model')\n\n#comparing accuracy(training)\npredictiondf.sort_values(by=['Accuracy(Training)'], ascending=False, inplace=True)\n\nsns.barplot(x='Accuracy(Training)', y='Model', data = predictiondf, palette='Blues_d', ax = ax[1][0])\nax[1][0].set_xlabel('Accuracy(Training)', size=16)\nax[1][0].set_ylabel('Model')\nax[1][0].set_xlim(0,1.0)\nax[1][0].set_xticks(np.arange(0, 1.1, 0.1))\n\n#comparing accuracy(testing)\npredictiondf.sort_values(by=['Accuracy(Test)'], ascending=False, inplace=True)\n\nsns.barplot(x='Accuracy(Test)', y='Model', data = predictiondf, palette='Reds_d', ax = ax[1][1])\nax[1][1].set_xlabel('Accuracy(Test)', size=16)\nax[1][1].set_ylabel('Model')\nax[1][1].set_xlim(0,1.0)\nax[1][1].set_xticks(np.arange(0, 1.1, 0.1))\n\nplt.show()","metadata":{"papermill":{"duration":0.508805,"end_time":"2022-07-04T08:10:33.533263","exception":false,"start_time":"2022-07-04T08:10:33.024458","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-04T18:59:40.007624Z","iopub.execute_input":"2022-07-04T18:59:40.008254Z","iopub.status.idle":"2022-07-04T18:59:40.811166Z","shell.execute_reply.started":"2022-07-04T18:59:40.008218Z","shell.execute_reply":"2022-07-04T18:59:40.809955Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Comparing how many false prediction made by each model\npredictiondf.sort_values(by=(['Accuracy(Test)']), ascending=True, inplace=True)\n\nf, axe = plt.subplots(1,1, figsize=(14,10))\nsns.barplot(x = predictiondf['Model'], y=predictiondf['False Positive'] + predictiondf['False Negative'], ax = axe)\naxe.set_xlabel('Model', size=16)\naxe.set_ylabel('False Observations', size=16)\n\nplt.show()","metadata":{"papermill":{"duration":0.309397,"end_time":"2022-07-04T08:10:33.86522","exception":false,"start_time":"2022-07-04T08:10:33.555823","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-04T18:59:40.813034Z","iopub.execute_input":"2022-07-04T18:59:40.813829Z","iopub.status.idle":"2022-07-04T18:59:41.046059Z","shell.execute_reply.started":"2022-07-04T18:59:40.813752Z","shell.execute_reply":"2022-07-04T18:59:41.04491Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Final Conclusion\n\nRandom Forest gives highest accuracy and thus gives lowest number of false observation. We should use Random Forest Classifier to predict the titanic data.","metadata":{"papermill":{"duration":0.033264,"end_time":"2022-07-04T08:10:33.921652","exception":false,"start_time":"2022-07-04T08:10:33.888388","status":"completed"},"tags":[]}}]}