{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n%matplotlib inline","metadata":{"execution":{"iopub.status.busy":"2022-02-02T16:16:46.275079Z","iopub.execute_input":"2022-02-02T16:16:46.275323Z","iopub.status.idle":"2022-02-02T16:16:46.280948Z","shell.execute_reply.started":"2022-02-02T16:16:46.275298Z","shell.execute_reply":"2022-02-02T16:16:46.280065Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"titanic_train_df = pd.read_csv('/kaggle/input/titanic/train.csv')","metadata":{"execution":{"iopub.status.busy":"2022-02-02T16:16:46.282856Z","iopub.execute_input":"2022-02-02T16:16:46.283321Z","iopub.status.idle":"2022-02-02T16:16:46.305579Z","shell.execute_reply.started":"2022-02-02T16:16:46.283288Z","shell.execute_reply":"2022-02-02T16:16:46.30485Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"titanic_train_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-02-02T16:16:46.306414Z","iopub.execute_input":"2022-02-02T16:16:46.307039Z","iopub.status.idle":"2022-02-02T16:16:46.324591Z","shell.execute_reply.started":"2022-02-02T16:16:46.307006Z","shell.execute_reply":"2022-02-02T16:16:46.324024Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Some notes about the dataframe columns:\n\nVariable\tDefinition\tKey\nsurvival\tSurvival\t0 = No, 1 = Yes\npclass\tTicket class\t1 = 1st, 2 = 2nd, 3 = 3rd\nsex\tSex\t\nAge\tAge in years\t\nsibsp\t# of siblings / spouses aboard the Titanic\t\nparch\t# of parents / children aboard the Titanic\t\nticket\tTicket number\t\nfare\tPassenger fare\t\ncabin\tCabin number\t\nembarked\tPort of Embarkation\tC = Cherbourg, Q = Queenstown, S = Southampton","metadata":{}},{"cell_type":"markdown","source":"<h1>Exploratory Data Analysis</h1>\n\n","metadata":{}},{"cell_type":"code","source":"sns.heatmap(titanic_train_df.isnull(),yticklabels=False,cbar=False)","metadata":{"execution":{"iopub.status.busy":"2022-02-02T16:16:46.325526Z","iopub.execute_input":"2022-02-02T16:16:46.326336Z","iopub.status.idle":"2022-02-02T16:16:46.54592Z","shell.execute_reply.started":"2022-02-02T16:16:46.326291Z","shell.execute_reply":"2022-02-02T16:16:46.545319Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<p>We found out a lot of the Cabin data is missing, and a significant Age data is missing.</p>\n<p>We might need to remove the entire cabin column as the imbalance might lead to over/under sampling</p>","metadata":{}},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"markdown","source":"<h2>Survived</h2>","metadata":{}},{"cell_type":"code","source":"\nsns.countplot(x='Survived',data=titanic_train_df)","metadata":{"execution":{"iopub.status.busy":"2022-02-02T16:16:46.548747Z","iopub.execute_input":"2022-02-02T16:16:46.548939Z","iopub.status.idle":"2022-02-02T16:16:46.671228Z","shell.execute_reply.started":"2022-02-02T16:16:46.548915Z","shell.execute_reply":"2022-02-02T16:16:46.670685Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<p>A quick plot count to glance the rough ratio of survivor and victim </p>\n    ","metadata":{}},{"cell_type":"code","source":"sns.countplot(x='Survived',data=titanic_train_df, hue='Sex')","metadata":{"execution":{"iopub.status.busy":"2022-02-02T16:16:46.674177Z","iopub.execute_input":"2022-02-02T16:16:46.675612Z","iopub.status.idle":"2022-02-02T16:16:46.825236Z","shell.execute_reply.started":"2022-02-02T16:16:46.67558Z","shell.execute_reply":"2022-02-02T16:16:46.820027Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<p>Separating survival data with genders by adding hue. </p>\n<br>\n<h2>Sex</h2>\n<p>Lets try seeing the initial passenger data based on gender proportion</p>","metadata":{}},{"cell_type":"code","source":"sns.countplot(x='Sex',data=titanic_train_df) ","metadata":{"execution":{"iopub.status.busy":"2022-02-02T16:16:46.828358Z","iopub.execute_input":"2022-02-02T16:16:46.828596Z","iopub.status.idle":"2022-02-02T16:16:46.941789Z","shell.execute_reply.started":"2022-02-02T16:16:46.828569Z","shell.execute_reply":"2022-02-02T16:16:46.941325Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.countplot(x='Sex',data=titanic_train_df, hue='Survived')","metadata":{"execution":{"iopub.status.busy":"2022-02-02T16:16:46.944619Z","iopub.execute_input":"2022-02-02T16:16:46.946057Z","iopub.status.idle":"2022-02-02T16:16:47.085544Z","shell.execute_reply.started":"2022-02-02T16:16:46.946026Z","shell.execute_reply":"2022-02-02T16:16:47.085021Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"\n<p>Adding hue to survival status, we see male passengers are way more likely to not survive</p>\n<br>\n<h2>Passenger Class</h2>\n<p>Lets try to separate them based on Passenger class data</p>","metadata":{}},{"cell_type":"code","source":"sns.countplot(x='Sex',data=titanic_train_df, hue='Pclass')","metadata":{"execution":{"iopub.status.busy":"2022-02-02T16:16:47.088554Z","iopub.execute_input":"2022-02-02T16:16:47.090092Z","iopub.status.idle":"2022-02-02T16:16:47.254559Z","shell.execute_reply.started":"2022-02-02T16:16:47.090057Z","shell.execute_reply":"2022-02-02T16:16:47.253316Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.countplot(x='Survived',data=titanic_train_df, hue='Pclass')","metadata":{"execution":{"iopub.status.busy":"2022-02-02T16:16:47.255608Z","iopub.execute_input":"2022-02-02T16:16:47.255902Z","iopub.status.idle":"2022-02-02T16:16:47.422035Z","shell.execute_reply.started":"2022-02-02T16:16:47.255881Z","shell.execute_reply":"2022-02-02T16:16:47.421004Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.countplot(x='Pclass',data=titanic_train_df, hue='Survived')","metadata":{"execution":{"iopub.status.busy":"2022-02-02T16:16:47.423306Z","iopub.execute_input":"2022-02-02T16:16:47.423472Z","iopub.status.idle":"2022-02-02T16:16:47.589062Z","shell.execute_reply.started":"2022-02-02T16:16:47.423451Z","shell.execute_reply":"2022-02-02T16:16:47.588288Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<p>Third Class passengers were more likely to die in the incident. However, it might be due to high proportion of male passengers there as well.</p>\n<p>First class passengers has most likelihood to survive. And the Third Class passengers has lowest likelihood of survival</p>\n<br>\n<p>Lets now see passengers age distribution data</p>","metadata":{}},{"cell_type":"code","source":"sns.displot(titanic_train_df['Age'].dropna(),kde=False,bins=30)","metadata":{"execution":{"iopub.status.busy":"2022-02-02T16:16:47.590246Z","iopub.execute_input":"2022-02-02T16:16:47.590418Z","iopub.status.idle":"2022-02-02T16:16:47.840095Z","shell.execute_reply.started":"2022-02-02T16:16:47.590398Z","shell.execute_reply":"2022-02-02T16:16:47.839384Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<p>We observe a Normal Distribution curve, with Mean in between 20-30.</p>\n","metadata":{}},{"cell_type":"code","source":"titanic_train_df.info()","metadata":{"execution":{"iopub.status.busy":"2022-02-02T16:16:47.84261Z","iopub.execute_input":"2022-02-02T16:16:47.842793Z","iopub.status.idle":"2022-02-02T16:16:47.858072Z","shell.execute_reply.started":"2022-02-02T16:16:47.842771Z","shell.execute_reply":"2022-02-02T16:16:47.857518Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.countplot(x=\"SibSp\",data=titanic_train_df)","metadata":{"execution":{"iopub.status.busy":"2022-02-02T16:16:47.859361Z","iopub.execute_input":"2022-02-02T16:16:47.859672Z","iopub.status.idle":"2022-02-02T16:16:48.032358Z","shell.execute_reply.started":"2022-02-02T16:16:47.859648Z","shell.execute_reply":"2022-02-02T16:16:48.031525Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<p>Most people in the Titanice has zero Sibling/Spouse, which suggested they might be solo traveler.</p>\n<p> I hypothesized the high number of 0 SibSp is probably related to the high level of males in the third class passengers.</p>\n\n<p>Lets try to dig deeper into the hypothesis</p>","metadata":{}},{"cell_type":"code","source":"sns.countplot(x=\"SibSp\",data=titanic_train_df,hue=\"Survived\")","metadata":{"execution":{"iopub.status.busy":"2022-02-02T16:16:48.033603Z","iopub.execute_input":"2022-02-02T16:16:48.03379Z","iopub.status.idle":"2022-02-02T16:16:48.242782Z","shell.execute_reply.started":"2022-02-02T16:16:48.033765Z","shell.execute_reply":"2022-02-02T16:16:48.241914Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<p>Small observation that people with only 1 spouse/siblings has higher survival ratio than other SibSp category</p>","metadata":{}},{"cell_type":"code","source":"sns.countplot(x=\"SibSp\",data=titanic_train_df,hue=\"Sex\")","metadata":{"execution":{"iopub.status.busy":"2022-02-02T16:16:48.243889Z","iopub.execute_input":"2022-02-02T16:16:48.244061Z","iopub.status.idle":"2022-02-02T16:16:48.550553Z","shell.execute_reply.started":"2022-02-02T16:16:48.24404Z","shell.execute_reply":"2022-02-02T16:16:48.550012Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.countplot(x=\"SibSp\",data=titanic_train_df,hue=\"Pclass\")","metadata":{"execution":{"iopub.status.busy":"2022-02-02T16:16:48.551558Z","iopub.execute_input":"2022-02-02T16:16:48.55212Z","iopub.status.idle":"2022-02-02T16:16:48.806677Z","shell.execute_reply.started":"2022-02-02T16:16:48.552094Z","shell.execute_reply":"2022-02-02T16:16:48.805884Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<p> The data above prove the hypothesis that the high number of 0 SibSp is probably related to the high level of males in the third class passengers.</p>\n<br>\n<p>Lets try to see distribution of fare people pay to get on board.</p>","metadata":{}},{"cell_type":"code","source":"\nsns.displot(titanic_train_df['Fare'].dropna(),kde=False,bins=50,height=8,aspect=1.8)","metadata":{"execution":{"iopub.status.busy":"2022-02-02T16:16:48.807742Z","iopub.execute_input":"2022-02-02T16:16:48.807944Z","iopub.status.idle":"2022-02-02T16:16:49.164775Z","shell.execute_reply.started":"2022-02-02T16:16:48.807918Z","shell.execute_reply":"2022-02-02T16:16:49.164002Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<p>Majority of Titanic passengers pay the cheapest fare, which falls under the first bin.</p>","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h1>Data Cleaning</h1>\n<br>\n<br>    \n<p>So firstly, we found out that a significant portion of age was missing from the data set. Instead of dropping all the missing ages, we can simply fill in the age with the mean age.</p>\n<p>Lets check the average age of different passenger class</p>","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(10,7))\nsns.boxplot(x='Pclass',y='Age',data=titanic_train_df)","metadata":{"execution":{"iopub.status.busy":"2022-02-02T16:16:49.165813Z","iopub.execute_input":"2022-02-02T16:16:49.16599Z","iopub.status.idle":"2022-02-02T16:16:49.33846Z","shell.execute_reply.started":"2022-02-02T16:16:49.165953Z","shell.execute_reply":"2022-02-02T16:16:49.337703Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<p>Each passenger class has different age spread, probably due to wealthier passengers were usually older.</p>\n<p>Now we are ready to impute by making a function</p>","metadata":{}},{"cell_type":"code","source":"pclass_age_mean = titanic_train_df.groupby('Pclass')['Age'].mean()\nprint(pclass_age_mean)","metadata":{"execution":{"iopub.status.busy":"2022-02-02T16:16:49.339504Z","iopub.execute_input":"2022-02-02T16:16:49.339746Z","iopub.status.idle":"2022-02-02T16:16:49.347995Z","shell.execute_reply.started":"2022-02-02T16:16:49.339714Z","shell.execute_reply":"2022-02-02T16:16:49.347094Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def impute_age(cols):\n    Age = cols[0] #1st column\n    Pclass=cols[1] #2nd column\n    if pd.isnull(Age):\n        if Pclass == 1:\n            return 38\n        elif Pclass == 2:\n            return 29\n        else:\n            return 24\n    else:\n        return Age","metadata":{"execution":{"iopub.status.busy":"2022-02-02T16:16:49.349619Z","iopub.execute_input":"2022-02-02T16:16:49.350267Z","iopub.status.idle":"2022-02-02T16:16:49.363625Z","shell.execute_reply.started":"2022-02-02T16:16:49.35023Z","shell.execute_reply":"2022-02-02T16:16:49.363Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"titanic_train_df['Age'] = titanic_train_df[['Age','Pclass']].apply(impute_age,axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-02-02T16:16:49.364719Z","iopub.execute_input":"2022-02-02T16:16:49.364914Z","iopub.status.idle":"2022-02-02T16:16:49.392516Z","shell.execute_reply.started":"2022-02-02T16:16:49.364887Z","shell.execute_reply":"2022-02-02T16:16:49.391684Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.heatmap(titanic_train_df.isnull(),yticklabels=False,cbar=False)","metadata":{"execution":{"iopub.status.busy":"2022-02-02T16:16:49.393859Z","iopub.execute_input":"2022-02-02T16:16:49.394131Z","iopub.status.idle":"2022-02-02T16:16:49.547091Z","shell.execute_reply.started":"2022-02-02T16:16:49.394105Z","shell.execute_reply":"2022-02-02T16:16:49.545805Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<p>The Age data has now been cleaned, as we had imputed the nulls with average age of respective passenger classes</p>\n<br>","metadata":{}},{"cell_type":"markdown","source":"<p>We will also write imputation method for Fare, as it is needed for the test data</p>\n<br>","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(10,7))\nsns.boxplot(x='Pclass',y='Fare',data=titanic_train_df)","metadata":{"execution":{"iopub.status.busy":"2022-02-02T16:16:49.548241Z","iopub.execute_input":"2022-02-02T16:16:49.548415Z","iopub.status.idle":"2022-02-02T16:16:49.710292Z","shell.execute_reply.started":"2022-02-02T16:16:49.548394Z","shell.execute_reply":"2022-02-02T16:16:49.709795Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pclass_fare_mean = titanic_train_df.groupby('Pclass')['Fare'].mean()\nprint(pclass_fare_mean)","metadata":{"execution":{"iopub.status.busy":"2022-02-02T16:16:49.71121Z","iopub.execute_input":"2022-02-02T16:16:49.711885Z","iopub.status.idle":"2022-02-02T16:16:49.719601Z","shell.execute_reply.started":"2022-02-02T16:16:49.711856Z","shell.execute_reply":"2022-02-02T16:16:49.718017Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def impute_fare(cols):\n    Fare = cols[0] #1st column\n    Pclass=cols[1] #2nd column\n    if pd.isnull(Fare):\n        if Pclass == 1:\n            return 84.154687\n        elif Pclass == 2:\n            return 20.662183\n        else:\n            return 13.675550\n    else:\n        return Fare","metadata":{"execution":{"iopub.status.busy":"2022-02-02T16:16:49.721263Z","iopub.execute_input":"2022-02-02T16:16:49.721431Z","iopub.status.idle":"2022-02-02T16:16:49.730361Z","shell.execute_reply.started":"2022-02-02T16:16:49.721411Z","shell.execute_reply":"2022-02-02T16:16:49.729824Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(12, 7))\nsns.heatmap(titanic_train_df.isnull(),yticklabels=False,cbar=False,cmap='viridis')","metadata":{"execution":{"iopub.status.busy":"2022-02-02T16:16:49.731287Z","iopub.execute_input":"2022-02-02T16:16:49.732083Z","iopub.status.idle":"2022-02-02T16:16:49.90017Z","shell.execute_reply.started":"2022-02-02T16:16:49.732048Z","shell.execute_reply":"2022-02-02T16:16:49.899516Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<p>Now we are left with the Cabin data and 1 Embarked row to be cleaned. However, as there is way too much missing Cabin data, I think its best to drop the data to avoid over/under sampling imbalance.</p>","metadata":{}},{"cell_type":"code","source":"titanic_train_df.drop('Cabin',axis=1,inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-02-02T16:16:49.901089Z","iopub.execute_input":"2022-02-02T16:16:49.901261Z","iopub.status.idle":"2022-02-02T16:16:49.906949Z","shell.execute_reply.started":"2022-02-02T16:16:49.901237Z","shell.execute_reply":"2022-02-02T16:16:49.90623Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Now we are left with one missing Embarked data. Lets drop the row as it is just one data point\n","metadata":{}},{"cell_type":"code","source":"titanic_train_df.dropna(inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-02-02T16:16:49.908047Z","iopub.execute_input":"2022-02-02T16:16:49.908507Z","iopub.status.idle":"2022-02-02T16:16:49.923873Z","shell.execute_reply.started":"2022-02-02T16:16:49.90847Z","shell.execute_reply":"2022-02-02T16:16:49.9231Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.heatmap(titanic_train_df.isnull(),yticklabels=False,cbar=False)","metadata":{"execution":{"iopub.status.busy":"2022-02-02T16:16:49.924898Z","iopub.execute_input":"2022-02-02T16:16:49.925146Z","iopub.status.idle":"2022-02-02T16:16:50.070888Z","shell.execute_reply.started":"2022-02-02T16:16:49.925112Z","shell.execute_reply":"2022-02-02T16:16:50.069544Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<p>Next up, we are going to set up dummy variables on categorical features. For example, making Sex category into boolean</p>\n","metadata":{}},{"cell_type":"code","source":"pd.get_dummies(titanic_train_df['Sex'])#pandas function","metadata":{"execution":{"iopub.status.busy":"2022-02-02T16:16:50.23416Z","iopub.execute_input":"2022-02-02T16:16:50.234348Z","iopub.status.idle":"2022-02-02T16:16:50.244594Z","shell.execute_reply.started":"2022-02-02T16:16:50.234326Z","shell.execute_reply":"2022-02-02T16:16:50.244181Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<p>One of the two columns above is redundant because we only need one column to identify sex. Having two columns will introduce multicollinearity, as the two columns are perfect predictors of each other.</p>","metadata":{}},{"cell_type":"code","source":"sex = pd.get_dummies(titanic_train_df['Sex'],drop_first=True)","metadata":{"execution":{"iopub.status.busy":"2022-02-02T16:16:50.245395Z","iopub.execute_input":"2022-02-02T16:16:50.245625Z","iopub.status.idle":"2022-02-02T16:16:50.262905Z","shell.execute_reply.started":"2022-02-02T16:16:50.245604Z","shell.execute_reply":"2022-02-02T16:16:50.262247Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sex.head()","metadata":{"execution":{"iopub.status.busy":"2022-02-02T16:16:50.264485Z","iopub.execute_input":"2022-02-02T16:16:50.26475Z","iopub.status.idle":"2022-02-02T16:16:50.283435Z","shell.execute_reply.started":"2022-02-02T16:16:50.264717Z","shell.execute_reply":"2022-02-02T16:16:50.282781Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"embark = pd.get_dummies(titanic_train_df['Embarked'],drop_first=True)","metadata":{"execution":{"iopub.status.busy":"2022-02-02T16:16:50.286516Z","iopub.execute_input":"2022-02-02T16:16:50.287259Z","iopub.status.idle":"2022-02-02T16:16:50.297956Z","shell.execute_reply.started":"2022-02-02T16:16:50.287226Z","shell.execute_reply":"2022-02-02T16:16:50.297421Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"embark.head()","metadata":{"execution":{"iopub.status.busy":"2022-02-02T16:16:50.303408Z","iopub.execute_input":"2022-02-02T16:16:50.304537Z","iopub.status.idle":"2022-02-02T16:16:50.315413Z","shell.execute_reply.started":"2022-02-02T16:16:50.30448Z","shell.execute_reply":"2022-02-02T16:16:50.314554Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.concat([titanic_train_df,sex,embark],axis=1) #we concatenate the new columns into the dataset","metadata":{"execution":{"iopub.status.busy":"2022-02-02T16:16:50.316495Z","iopub.execute_input":"2022-02-02T16:16:50.316756Z","iopub.status.idle":"2022-02-02T16:16:50.326023Z","shell.execute_reply.started":"2022-02-02T16:16:50.316723Z","shell.execute_reply":"2022-02-02T16:16:50.325599Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head()","metadata":{"execution":{"iopub.status.busy":"2022-02-02T16:16:50.327118Z","iopub.execute_input":"2022-02-02T16:16:50.327947Z","iopub.status.idle":"2022-02-02T16:16:50.350642Z","shell.execute_reply.started":"2022-02-02T16:16:50.327898Z","shell.execute_reply":"2022-02-02T16:16:50.35001Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.drop(['Sex','Embarked','Name','Ticket','PassengerId'],axis=1,inplace=True)#we can now drop the Sex and Embarked columns. \n           #We will also drop the Name and Ticketas we wont be using it for machine learning algorithm\n ","metadata":{"execution":{"iopub.status.busy":"2022-02-02T16:16:50.351621Z","iopub.execute_input":"2022-02-02T16:16:50.352082Z","iopub.status.idle":"2022-02-02T16:16:50.365237Z","shell.execute_reply.started":"2022-02-02T16:16:50.352052Z","shell.execute_reply":"2022-02-02T16:16:50.364442Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head()","metadata":{"execution":{"iopub.status.busy":"2022-02-02T16:16:50.366715Z","iopub.execute_input":"2022-02-02T16:16:50.368373Z","iopub.status.idle":"2022-02-02T16:16:50.383657Z","shell.execute_reply.started":"2022-02-02T16:16:50.368339Z","shell.execute_reply":"2022-02-02T16:16:50.382562Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"corr=train.corr().round(1)\n\nplt.figure(figsize=(25, 20))\nsns.set(style=\"ticks\", context=\"talk\",font_scale = 2)\nplt.style.use(\"dark_background\")\nmask = np.zeros_like(corr)\nmask[np.triu_indices_from(mask)] = True\nsns.heatmap(corr,annot=True,cmap='Purples',mask=mask,cbar=True)\nplt.title('Correlation Plot')","metadata":{"execution":{"iopub.status.busy":"2022-02-02T16:16:50.384724Z","iopub.execute_input":"2022-02-02T16:16:50.384895Z","iopub.status.idle":"2022-02-02T16:16:50.961714Z","shell.execute_reply.started":"2022-02-02T16:16:50.384873Z","shell.execute_reply":"2022-02-02T16:16:50.949513Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<p>We also need to convert Pclass into dummy variables as well. We will do that later</p>","metadata":{}},{"cell_type":"code","source":"X= train.drop('Survived',axis=1)\ny=train['Survived']","metadata":{"execution":{"iopub.status.busy":"2022-02-02T16:16:50.962795Z","iopub.execute_input":"2022-02-02T16:16:50.963344Z","iopub.status.idle":"2022-02-02T16:16:50.967543Z","shell.execute_reply.started":"2022-02-02T16:16:50.963319Z","shell.execute_reply":"2022-02-02T16:16:50.967007Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split","metadata":{"execution":{"iopub.status.busy":"2022-02-02T16:16:50.968413Z","iopub.execute_input":"2022-02-02T16:16:50.968975Z","iopub.status.idle":"2022-02-02T16:16:51.125183Z","shell.execute_reply.started":"2022-02-02T16:16:50.968934Z","shell.execute_reply":"2022-02-02T16:16:51.124493Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.3, random_state=101)","metadata":{"execution":{"iopub.status.busy":"2022-02-02T16:16:51.126142Z","iopub.execute_input":"2022-02-02T16:16:51.126329Z","iopub.status.idle":"2022-02-02T16:16:51.131633Z","shell.execute_reply.started":"2022-02-02T16:16:51.126306Z","shell.execute_reply":"2022-02-02T16:16:51.131017Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.linear_model import LogisticRegression","metadata":{"execution":{"iopub.status.busy":"2022-02-02T16:16:51.132799Z","iopub.execute_input":"2022-02-02T16:16:51.133204Z","iopub.status.idle":"2022-02-02T16:16:51.214783Z","shell.execute_reply.started":"2022-02-02T16:16:51.13316Z","shell.execute_reply":"2022-02-02T16:16:51.214314Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"logmodel = LogisticRegression(max_iter=10000)","metadata":{"execution":{"iopub.status.busy":"2022-02-02T16:16:51.215831Z","iopub.execute_input":"2022-02-02T16:16:51.216222Z","iopub.status.idle":"2022-02-02T16:16:51.220293Z","shell.execute_reply.started":"2022-02-02T16:16:51.216186Z","shell.execute_reply":"2022-02-02T16:16:51.21919Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"logmodel.fit(X_train,y_train)","metadata":{"execution":{"iopub.status.busy":"2022-02-02T16:16:51.221511Z","iopub.execute_input":"2022-02-02T16:16:51.221855Z","iopub.status.idle":"2022-02-02T16:16:51.282256Z","shell.execute_reply.started":"2022-02-02T16:16:51.221822Z","shell.execute_reply":"2022-02-02T16:16:51.281614Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions = logmodel.predict(X_test)","metadata":{"execution":{"iopub.status.busy":"2022-02-02T16:16:51.283036Z","iopub.execute_input":"2022-02-02T16:16:51.283641Z","iopub.status.idle":"2022-02-02T16:16:51.288672Z","shell.execute_reply.started":"2022-02-02T16:16:51.283615Z","shell.execute_reply":"2022-02-02T16:16:51.287823Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import classification_report","metadata":{"execution":{"iopub.status.busy":"2022-02-02T16:16:51.291673Z","iopub.execute_input":"2022-02-02T16:16:51.292502Z","iopub.status.idle":"2022-02-02T16:16:51.301473Z","shell.execute_reply.started":"2022-02-02T16:16:51.292468Z","shell.execute_reply":"2022-02-02T16:16:51.300507Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(classification_report(y_test,predictions))","metadata":{"execution":{"iopub.status.busy":"2022-02-02T16:16:51.304348Z","iopub.execute_input":"2022-02-02T16:16:51.305882Z","iopub.status.idle":"2022-02-02T16:16:51.319434Z","shell.execute_reply.started":"2022-02-02T16:16:51.305848Z","shell.execute_reply":"2022-02-02T16:16:51.318724Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import confusion_matrix","metadata":{"execution":{"iopub.status.busy":"2022-02-02T16:16:51.320658Z","iopub.execute_input":"2022-02-02T16:16:51.321053Z","iopub.status.idle":"2022-02-02T16:16:51.328899Z","shell.execute_reply.started":"2022-02-02T16:16:51.32103Z","shell.execute_reply":"2022-02-02T16:16:51.328273Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"confusion_matrix(y_test,predictions)","metadata":{"execution":{"iopub.status.busy":"2022-02-02T16:16:51.329941Z","iopub.execute_input":"2022-02-02T16:16:51.330882Z","iopub.status.idle":"2022-02-02T16:16:51.346816Z","shell.execute_reply.started":"2022-02-02T16:16:51.330829Z","shell.execute_reply":"2022-02-02T16:16:51.345659Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions","metadata":{"execution":{"iopub.status.busy":"2022-02-02T16:16:51.347779Z","iopub.execute_input":"2022-02-02T16:16:51.347983Z","iopub.status.idle":"2022-02-02T16:16:51.361997Z","shell.execute_reply.started":"2022-02-02T16:16:51.347943Z","shell.execute_reply":"2022-02-02T16:16:51.361492Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"\nTime to get the real test dataset csv file from. [Titanic Data Set from Kaggle](https://www.kaggle.com/c/titanic).","metadata":{}},{"cell_type":"markdown","source":"<h1>Evaluation</h1>","metadata":{}},{"cell_type":"code","source":"from sklearn.metrics import classification_report","metadata":{"execution":{"iopub.status.busy":"2022-02-02T16:16:51.362707Z","iopub.execute_input":"2022-02-02T16:16:51.363891Z","iopub.status.idle":"2022-02-02T16:16:51.375029Z","shell.execute_reply.started":"2022-02-02T16:16:51.363846Z","shell.execute_reply":"2022-02-02T16:16:51.373779Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(classification_report(y_test,predictions))","metadata":{"execution":{"iopub.status.busy":"2022-02-02T16:16:51.375903Z","iopub.execute_input":"2022-02-02T16:16:51.376682Z","iopub.status.idle":"2022-02-02T16:16:51.393462Z","shell.execute_reply.started":"2022-02-02T16:16:51.376637Z","shell.execute_reply":"2022-02-02T16:16:51.392021Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"titanic_test_df = pd.read_csv('/kaggle/input/titanic/test.csv') #test.csv from kaggle","metadata":{"execution":{"iopub.status.busy":"2022-02-02T16:18:28.028555Z","iopub.execute_input":"2022-02-02T16:18:28.028809Z","iopub.status.idle":"2022-02-02T16:18:28.0422Z","shell.execute_reply.started":"2022-02-02T16:18:28.028777Z","shell.execute_reply":"2022-02-02T16:18:28.04174Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.heatmap(titanic_test_df.isnull(),yticklabels=False,cbar=False)","metadata":{"execution":{"iopub.status.busy":"2022-02-02T16:18:30.619868Z","iopub.execute_input":"2022-02-02T16:18:30.620828Z","iopub.status.idle":"2022-02-02T16:18:30.716257Z","shell.execute_reply.started":"2022-02-02T16:18:30.6208Z","shell.execute_reply":"2022-02-02T16:18:30.715782Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"titanic_test_df['Age'] = titanic_test_df[['Age','Pclass']].apply(impute_age,axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-02-02T16:18:33.235112Z","iopub.execute_input":"2022-02-02T16:18:33.236336Z","iopub.status.idle":"2022-02-02T16:18:33.246126Z","shell.execute_reply.started":"2022-02-02T16:18:33.236289Z","shell.execute_reply":"2022-02-02T16:18:33.245431Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.heatmap(titanic_test_df.isnull(),yticklabels=False,cbar=False)","metadata":{"execution":{"iopub.status.busy":"2022-02-02T16:18:34.766023Z","iopub.execute_input":"2022-02-02T16:18:34.766244Z","iopub.status.idle":"2022-02-02T16:18:34.867655Z","shell.execute_reply.started":"2022-02-02T16:18:34.766222Z","shell.execute_reply":"2022-02-02T16:18:34.866863Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We will impute the missing Fare information using the average respective Pclass data","metadata":{}},{"cell_type":"code","source":"titanic_test_df['Fare'] = titanic_test_df[['Fare','Pclass']].apply(impute_fare,axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-02-02T16:18:39.238302Z","iopub.execute_input":"2022-02-02T16:18:39.238535Z","iopub.status.idle":"2022-02-02T16:18:39.248259Z","shell.execute_reply.started":"2022-02-02T16:18:39.238504Z","shell.execute_reply":"2022-02-02T16:18:39.247809Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.heatmap(titanic_test_df.isnull(),yticklabels=False,cbar=False)","metadata":{"execution":{"iopub.status.busy":"2022-02-02T16:18:42.352508Z","iopub.execute_input":"2022-02-02T16:18:42.352851Z","iopub.status.idle":"2022-02-02T16:18:42.468495Z","shell.execute_reply.started":"2022-02-02T16:18:42.352823Z","shell.execute_reply":"2022-02-02T16:18:42.468007Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sex = pd.get_dummies(titanic_test_df['Sex'],drop_first=True)","metadata":{"execution":{"iopub.status.busy":"2022-02-02T16:18:44.632299Z","iopub.execute_input":"2022-02-02T16:18:44.632632Z","iopub.status.idle":"2022-02-02T16:18:44.63737Z","shell.execute_reply.started":"2022-02-02T16:18:44.632597Z","shell.execute_reply":"2022-02-02T16:18:44.636512Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"embark = pd.get_dummies(titanic_test_df['Embarked'],drop_first=True)","metadata":{"execution":{"iopub.status.busy":"2022-02-02T16:18:46.026334Z","iopub.execute_input":"2022-02-02T16:18:46.027235Z","iopub.status.idle":"2022-02-02T16:18:46.030849Z","shell.execute_reply.started":"2022-02-02T16:18:46.027209Z","shell.execute_reply":"2022-02-02T16:18:46.030463Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test = pd.concat([titanic_test_df,sex,embark],axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-02-02T16:18:47.428503Z","iopub.execute_input":"2022-02-02T16:18:47.428918Z","iopub.status.idle":"2022-02-02T16:18:47.434524Z","shell.execute_reply.started":"2022-02-02T16:18:47.428895Z","shell.execute_reply":"2022-02-02T16:18:47.43275Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.head()","metadata":{"execution":{"iopub.status.busy":"2022-02-02T16:18:49.191881Z","iopub.execute_input":"2022-02-02T16:18:49.193005Z","iopub.status.idle":"2022-02-02T16:18:49.205392Z","shell.execute_reply.started":"2022-02-02T16:18:49.192919Z","shell.execute_reply":"2022-02-02T16:18:49.20486Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.drop(['Sex','Embarked','Name','Ticket','PassengerId','Cabin'],axis=1,inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-02-02T16:18:51.413216Z","iopub.execute_input":"2022-02-02T16:18:51.413749Z","iopub.status.idle":"2022-02-02T16:18:51.419142Z","shell.execute_reply.started":"2022-02-02T16:18:51.413719Z","shell.execute_reply":"2022-02-02T16:18:51.418561Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.head()","metadata":{"execution":{"iopub.status.busy":"2022-02-02T16:18:52.792794Z","iopub.execute_input":"2022-02-02T16:18:52.793334Z","iopub.status.idle":"2022-02-02T16:18:52.807223Z","shell.execute_reply.started":"2022-02-02T16:18:52.793298Z","shell.execute_reply":"2022-02-02T16:18:52.806179Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.heatmap(test.isnull(),yticklabels=False,cbar=False)","metadata":{"execution":{"iopub.status.busy":"2022-02-02T16:18:54.3007Z","iopub.execute_input":"2022-02-02T16:18:54.300914Z","iopub.status.idle":"2022-02-02T16:18:54.418243Z","shell.execute_reply.started":"2022-02-02T16:18:54.300891Z","shell.execute_reply":"2022-02-02T16:18:54.417418Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions = logmodel.predict(test)","metadata":{"execution":{"iopub.status.busy":"2022-02-02T16:18:56.988471Z","iopub.execute_input":"2022-02-02T16:18:56.989379Z","iopub.status.idle":"2022-02-02T16:18:56.994618Z","shell.execute_reply.started":"2022-02-02T16:18:56.989346Z","shell.execute_reply":"2022-02-02T16:18:56.994152Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions","metadata":{"execution":{"iopub.status.busy":"2022-02-02T16:18:59.954005Z","iopub.execute_input":"2022-02-02T16:18:59.954392Z","iopub.status.idle":"2022-02-02T16:18:59.961551Z","shell.execute_reply.started":"2022-02-02T16:18:59.954352Z","shell.execute_reply":"2022-02-02T16:18:59.96033Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"export = pd.DataFrame({'PassengerId':titanic_test_df.PassengerId, 'Survived': predictions})\nexport.to_csv('submission.csv', index=False)\nexport.head()","metadata":{"execution":{"iopub.status.busy":"2022-02-02T16:19:08.190268Z","iopub.execute_input":"2022-02-02T16:19:08.190684Z","iopub.status.idle":"2022-02-02T16:19:08.202509Z","shell.execute_reply.started":"2022-02-02T16:19:08.190658Z","shell.execute_reply":"2022-02-02T16:19:08.201603Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Things to improve in the future:\n1. Use categorical data of Pclass. (use dummy variables instead of 1,2,3).\n2. Remove Embarked dummy variables and replace with categorical Pclass dummy variables","metadata":{}}]}