{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-27T15:46:44.542888Z","iopub.execute_input":"2022-07-27T15:46:44.543517Z","iopub.status.idle":"2022-07-27T15:46:44.566750Z","shell.execute_reply.started":"2022-07-27T15:46:44.543395Z","shell.execute_reply":"2022-07-27T15:46:44.565255Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import accuracy_score, classification_report, confusion_matrix\nfrom sklearn.preprocessing import StandardScaler, LabelEncoder\nfrom sklearn.ensemble import RandomForestClassifier, GradientBoostingClassifier\nfrom sklearn.tree import DecisionTreeClassifier\nfrom sklearn.linear_model import LogisticRegression\nfrom xgboost import XGBClassifier\nfrom sklearn.naive_bayes import GaussianNB\nfrom sklearn.neighbors import KNeighborsClassifier\nimport warnings\nwarnings.filterwarnings(\"ignore\")","metadata":{"execution":{"iopub.status.busy":"2022-07-27T15:46:44.568815Z","iopub.execute_input":"2022-07-27T15:46:44.569202Z","iopub.status.idle":"2022-07-27T15:46:45.423094Z","shell.execute_reply.started":"2022-07-27T15:46:44.569169Z","shell.execute_reply":"2022-07-27T15:46:45.421485Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Training Data","metadata":{}},{"cell_type":"markdown","source":"**Load the data**","metadata":{}},{"cell_type":"code","source":"data = pd.read_csv(\"../input/titanic/train.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-07-27T15:46:45.425780Z","iopub.execute_input":"2022-07-27T15:46:45.426673Z","iopub.status.idle":"2022-07-27T15:46:45.443044Z","shell.execute_reply.started":"2022-07-27T15:46:45.426603Z","shell.execute_reply":"2022-07-27T15:46:45.441373Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-27T15:46:45.446359Z","iopub.execute_input":"2022-07-27T15:46:45.447246Z","iopub.status.idle":"2022-07-27T15:46:45.472545Z","shell.execute_reply.started":"2022-07-27T15:46:45.447187Z","shell.execute_reply":"2022-07-27T15:46:45.471316Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# removing two unnecessary columns\ndata = data.drop(['PassengerId','Name'], axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-07-27T15:46:45.475334Z","iopub.execute_input":"2022-07-27T15:46:45.476086Z","iopub.status.idle":"2022-07-27T15:46:45.482746Z","shell.execute_reply.started":"2022-07-27T15:46:45.476046Z","shell.execute_reply":"2022-07-27T15:46:45.481547Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-27T15:46:45.484218Z","iopub.execute_input":"2022-07-27T15:46:45.484953Z","iopub.status.idle":"2022-07-27T15:46:45.499281Z","shell.execute_reply.started":"2022-07-27T15:46:45.484913Z","shell.execute_reply":"2022-07-27T15:46:45.497869Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-27T15:46:45.501025Z","iopub.execute_input":"2022-07-27T15:46:45.502163Z","iopub.status.idle":"2022-07-27T15:46:45.523333Z","shell.execute_reply.started":"2022-07-27T15:46:45.502120Z","shell.execute_reply":"2022-07-27T15:46:45.521746Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data.describe(include='all')","metadata":{"execution":{"iopub.status.busy":"2022-07-27T15:46:45.524881Z","iopub.execute_input":"2022-07-27T15:46:45.526013Z","iopub.status.idle":"2022-07-27T15:46:45.578415Z","shell.execute_reply.started":"2022-07-27T15:46:45.525942Z","shell.execute_reply":"2022-07-27T15:46:45.576919Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Dealing with null values**","metadata":{}},{"cell_type":"code","source":"data.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2022-07-27T15:46:45.579734Z","iopub.execute_input":"2022-07-27T15:46:45.580085Z","iopub.status.idle":"2022-07-27T15:46:45.592829Z","shell.execute_reply.started":"2022-07-27T15:46:45.580055Z","shell.execute_reply":"2022-07-27T15:46:45.591181Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Out of 891 records 687 values are missing in Cabin column, which is more that 75% missing data in a column. So dropping this column.","metadata":{}},{"cell_type":"code","source":"data = data.drop('Cabin', axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-07-27T15:46:45.595233Z","iopub.execute_input":"2022-07-27T15:46:45.596475Z","iopub.status.idle":"2022-07-27T15:46:45.605761Z","shell.execute_reply.started":"2022-07-27T15:46:45.596416Z","shell.execute_reply":"2022-07-27T15:46:45.604182Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"For the other two columns which have missing values - 'Age' and 'Embarked', Age column is continuous variable and Embarked is categorical. So replacing the null values of Age column with mean value and Embarked with mode.","metadata":{}},{"cell_type":"code","source":"data[['Embarked']] = pd.DataFrame(SimpleImputer(strategy='most_frequent').fit_transform(data[['Embarked']]))\ndata[['Age']] = pd.DataFrame(SimpleImputer(strategy='mean').fit_transform(data[['Age']]))","metadata":{"execution":{"iopub.status.busy":"2022-07-27T15:46:45.612320Z","iopub.execute_input":"2022-07-27T15:46:45.612916Z","iopub.status.idle":"2022-07-27T15:46:45.638020Z","shell.execute_reply.started":"2022-07-27T15:46:45.612832Z","shell.execute_reply":"2022-07-27T15:46:45.635663Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2022-07-27T15:46:45.643580Z","iopub.execute_input":"2022-07-27T15:46:45.644631Z","iopub.status.idle":"2022-07-27T15:46:45.656621Z","shell.execute_reply.started":"2022-07-27T15:46:45.644542Z","shell.execute_reply":"2022-07-27T15:46:45.654947Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Null values removed sucessfully.","metadata":{}},{"cell_type":"markdown","source":"**Unique values for each column**","metadata":{}},{"cell_type":"code","source":"for col in data.columns:\n    uni = data[col].unique().tolist()\n    print(f\"The no. of unique values of column '{col}' is: {len(uni)}\")","metadata":{"execution":{"iopub.status.busy":"2022-07-27T15:46:45.659125Z","iopub.execute_input":"2022-07-27T15:46:45.659525Z","iopub.status.idle":"2022-07-27T15:46:45.672900Z","shell.execute_reply.started":"2022-07-27T15:46:45.659491Z","shell.execute_reply":"2022-07-27T15:46:45.669800Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We have 681 different values in 'Ticket' column and we can't derive much from that column, so dropping it.","metadata":{}},{"cell_type":"code","source":"data = data.drop(['Ticket'], axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-07-27T15:46:45.674898Z","iopub.execute_input":"2022-07-27T15:46:45.676523Z","iopub.status.idle":"2022-07-27T15:46:45.687530Z","shell.execute_reply.started":"2022-07-27T15:46:45.676464Z","shell.execute_reply":"2022-07-27T15:46:45.684932Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Column with continous data**","metadata":{}},{"cell_type":"markdown","source":"We have two columns in our dataset which contain continuous data - 'Age' and 'Fare'.","metadata":{}},{"cell_type":"code","source":"data['Age'].describe()","metadata":{"execution":{"iopub.status.busy":"2022-07-27T15:46:45.689644Z","iopub.execute_input":"2022-07-27T15:46:45.691213Z","iopub.status.idle":"2022-07-27T15:46:45.708711Z","shell.execute_reply.started":"2022-07-27T15:46:45.691150Z","shell.execute_reply":"2022-07-27T15:46:45.707053Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data['Fare'].describe()","metadata":{"execution":{"iopub.status.busy":"2022-07-27T15:46:45.710508Z","iopub.execute_input":"2022-07-27T15:46:45.711027Z","iopub.status.idle":"2022-07-27T15:46:45.727607Z","shell.execute_reply.started":"2022-07-27T15:46:45.710977Z","shell.execute_reply":"2022-07-27T15:46:45.726501Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Let's visualise these two variables.","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(14,4))\nplt.subplot(1,2,1)\nsns.distplot(a=data['Age'])\nplt.subplot(1,2,2)\nsns.distplot(a=data['Fare'])\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-27T15:46:45.729047Z","iopub.execute_input":"2022-07-27T15:46:45.729424Z","iopub.status.idle":"2022-07-27T15:46:46.249855Z","shell.execute_reply.started":"2022-07-27T15:46:45.729390Z","shell.execute_reply":"2022-07-27T15:46:46.248497Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(14,4))\nplt.subplot(1,2,1)\nsns.violinplot(data['Age'])\nplt.subplot(1,2,2)\nsns.violinplot(data['Fare'])\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-27T15:46:46.251836Z","iopub.execute_input":"2022-07-27T15:46:46.252283Z","iopub.status.idle":"2022-07-27T15:46:46.538206Z","shell.execute_reply.started":"2022-07-27T15:46:46.252246Z","shell.execute_reply":"2022-07-27T15:46:46.536969Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Let us check for outliers for these two.","metadata":{}},{"cell_type":"code","source":"data[['Age','Fare']].boxplot()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-27T15:46:46.539752Z","iopub.execute_input":"2022-07-27T15:46:46.540151Z","iopub.status.idle":"2022-07-27T15:46:46.740489Z","shell.execute_reply.started":"2022-07-27T15:46:46.540116Z","shell.execute_reply":"2022-07-27T15:46:46.739540Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Both colum has outliers it seems. But here, in case of 'Age', the outliers simply mean aged/old persons. So we should not remove them.","metadata":{}},{"cell_type":"markdown","source":"Let's see the fare range of the passengers with respect to survival.","metadata":{}},{"cell_type":"code","source":"d1 = data[data['Survived']==1]['Fare'].tolist()\nd2 = data[data['Survived']==0]['Fare'].tolist()\nplt.figure(figsize=(8,6))\nplt.plot(d1, color='green')\nplt.plot(d2, color='red')\nplt.legend(labels=['Survived','Not Survived'])\nplt.title('Fare range on the basis of survival')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-27T15:46:46.741837Z","iopub.execute_input":"2022-07-27T15:46:46.742499Z","iopub.status.idle":"2022-07-27T15:46:46.996417Z","shell.execute_reply.started":"2022-07-27T15:46:46.742461Z","shell.execute_reply":"2022-07-27T15:46:46.995055Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Now instead of removing outliers from this dataset let us divide these two columns data in certain categories.","metadata":{}},{"cell_type":"code","source":"# dividing into age groups\nage_cat = []\nage = data['Age']\nfor i in age:\n    if i<13:\n        age_cat.append('Children')\n    elif i<20:\n        age_cat.append('Teenagers')\n    elif i<40:\n        age_cat.append('Adults')\n    elif i<60:\n        age_cat.append('Middle Age')\n    else:\n        age_cat.append('Senior Citizen')\n\ndata['Age Category'] = age_cat","metadata":{"execution":{"iopub.status.busy":"2022-07-27T15:46:46.998436Z","iopub.execute_input":"2022-07-27T15:46:46.998944Z","iopub.status.idle":"2022-07-27T15:46:47.011378Z","shell.execute_reply.started":"2022-07-27T15:46:46.998898Z","shell.execute_reply":"2022-07-27T15:46:47.009664Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# dividing into fare categories\nfare_category = []\nfare = data['Fare']\nfor f in fare:\n    if f<100:\n        fare_category.append('Low')\n    elif f<250:\n        fare_category.append('Average')\n    else:\n        fare_category.append('High')\ndata['Fare Category'] = fare_category\n","metadata":{"execution":{"iopub.status.busy":"2022-07-27T15:46:47.013601Z","iopub.execute_input":"2022-07-27T15:46:47.014190Z","iopub.status.idle":"2022-07-27T15:46:47.026131Z","shell.execute_reply.started":"2022-07-27T15:46:47.014136Z","shell.execute_reply":"2022-07-27T15:46:47.025146Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"So we added two new columns with categorical data derived from numerical columns.","metadata":{}},{"cell_type":"markdown","source":"**Dealing with discreet and categorical columns**","metadata":{}},{"cell_type":"markdown","source":"We have 'Pclass', 'SibSp', 'Parch'- these columns with numerical but discreet data. And 'Sex', 'Age Category', 'Embarked', 'Fare Category' as categorical data.","metadata":{}},{"cell_type":"markdown","source":"The column 'SibSp' means *no. of Siblings/Spouse*.\nThe column 'Parch' means *no. of Parent/Children*.\n*So we are adding these two columns as 'Family Members'.*","metadata":{}},{"cell_type":"code","source":"data['Family Members'] = data['SibSp'] + data['Parch']","metadata":{"execution":{"iopub.status.busy":"2022-07-27T15:46:47.028794Z","iopub.execute_input":"2022-07-27T15:46:47.029340Z","iopub.status.idle":"2022-07-27T15:46:47.041056Z","shell.execute_reply.started":"2022-07-27T15:46:47.029263Z","shell.execute_reply":"2022-07-27T15:46:47.040009Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# all the categorical columns \n# here dealing the discreet data also as categorical data just for EDA purpose\ncategorical = ['Pclass', 'Sex', 'SibSp', 'Parch', 'Embarked', 'Age Category', 'Family Members', 'Fare Category']","metadata":{"execution":{"iopub.status.busy":"2022-07-27T15:46:47.042277Z","iopub.execute_input":"2022-07-27T15:46:47.043183Z","iopub.status.idle":"2022-07-27T15:46:47.055580Z","shell.execute_reply.started":"2022-07-27T15:46:47.043142Z","shell.execute_reply":"2022-07-27T15:46:47.054587Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"For this purpose I am setting data types of these columns as string, but I do not want to change anything in the main dataset, so taking a copy of that in this case. ","metadata":{}},{"cell_type":"code","source":"df = data.copy()","metadata":{"execution":{"iopub.status.busy":"2022-07-27T15:46:47.056923Z","iopub.execute_input":"2022-07-27T15:46:47.057534Z","iopub.status.idle":"2022-07-27T15:46:47.068802Z","shell.execute_reply.started":"2022-07-27T15:46:47.057498Z","shell.execute_reply":"2022-07-27T15:46:47.067473Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# changing data type\nfor col in categorical:\n    df[col]=df[col].astype(str)","metadata":{"execution":{"iopub.status.busy":"2022-07-27T15:46:47.070266Z","iopub.execute_input":"2022-07-27T15:46:47.070809Z","iopub.status.idle":"2022-07-27T15:46:47.087463Z","shell.execute_reply.started":"2022-07-27T15:46:47.070776Z","shell.execute_reply":"2022-07-27T15:46:47.086157Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Creating a function which can take a column as input and will return the survival rate in percentage with respect to each category of that column.","metadata":{}},{"cell_type":"code","source":"def percentage(col):\n    unique = df[col].unique().tolist()\n    whole = df[col].tolist()\n    dic = {}\n    for j in unique:\n        total = whole.count(j)\n        surv = df.loc[(df[col]==j) & (df[\"Survived\"]==1)].shape[0]\n        perc = round(surv/total*100,2)\n        dic[j] = perc\n    return dic","metadata":{"execution":{"iopub.status.busy":"2022-07-27T15:46:47.088999Z","iopub.execute_input":"2022-07-27T15:46:47.089559Z","iopub.status.idle":"2022-07-27T15:46:47.097943Z","shell.execute_reply.started":"2022-07-27T15:46:47.089525Z","shell.execute_reply":"2022-07-27T15:46:47.096264Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# creating another function for visualisation\ndef visual(col):\n    plt.figure(figsize=(12,4))\n    plt.subplot(1,2,1)\n    sns.countplot(data=df, x=col, hue='Survived', palette='viridis')\n    plt.title(\"Survived and Not Survived count\")\n    plt.subplot(1,2,2)\n    plt.bar(x=percentage(col).keys(), height=percentage(col).values(), color='teal',alpha=0.6,edgecolor='black')\n    plt.title(f\"Survival rate w.r.t. each category of column '{col}'\")","metadata":{"execution":{"iopub.status.busy":"2022-07-27T15:46:47.099448Z","iopub.execute_input":"2022-07-27T15:46:47.100029Z","iopub.status.idle":"2022-07-27T15:46:47.111560Z","shell.execute_reply.started":"2022-07-27T15:46:47.099993Z","shell.execute_reply":"2022-07-27T15:46:47.109470Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Visualisation and analysis for each column**","metadata":{}},{"cell_type":"markdown","source":"**Column - 'Pclass'**","metadata":{}},{"cell_type":"code","source":"sns.countplot(df['Pclass'])\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-27T15:46:47.120195Z","iopub.execute_input":"2022-07-27T15:46:47.120632Z","iopub.status.idle":"2022-07-27T15:46:47.298968Z","shell.execute_reply.started":"2022-07-27T15:46:47.120595Z","shell.execute_reply":"2022-07-27T15:46:47.297810Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"percentage('Pclass')","metadata":{"execution":{"iopub.status.busy":"2022-07-27T15:46:47.300624Z","iopub.execute_input":"2022-07-27T15:46:47.301156Z","iopub.status.idle":"2022-07-27T15:46:47.317581Z","shell.execute_reply.started":"2022-07-27T15:46:47.301104Z","shell.execute_reply":"2022-07-27T15:46:47.316124Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"visual('Pclass')","metadata":{"execution":{"iopub.status.busy":"2022-07-27T15:46:47.319388Z","iopub.execute_input":"2022-07-27T15:46:47.320726Z","iopub.status.idle":"2022-07-27T15:46:47.655181Z","shell.execute_reply.started":"2022-07-27T15:46:47.320673Z","shell.execute_reply":"2022-07-27T15:46:47.653873Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Insights:**\n1. Lower class has the highest number of passengers.\n2. For lower class death rate is much higher than survival rate, for middle class both rates are almost same, for higher class survival rate is higher than death rate.\n3. Higher class has the highest survival rate (almost 63%) and class 1 has the lowest survival rate (approx 25%).","metadata":{}},{"cell_type":"markdown","source":"**Column - 'Sex'**","metadata":{}},{"cell_type":"code","source":"sns.countplot(df['Sex'])\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-27T15:46:47.656783Z","iopub.execute_input":"2022-07-27T15:46:47.658159Z","iopub.status.idle":"2022-07-27T15:46:47.833731Z","shell.execute_reply.started":"2022-07-27T15:46:47.658099Z","shell.execute_reply":"2022-07-27T15:46:47.832444Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"percentage('Sex')","metadata":{"execution":{"iopub.status.busy":"2022-07-27T15:46:47.835218Z","iopub.execute_input":"2022-07-27T15:46:47.835605Z","iopub.status.idle":"2022-07-27T15:46:47.849812Z","shell.execute_reply.started":"2022-07-27T15:46:47.835570Z","shell.execute_reply":"2022-07-27T15:46:47.848378Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"visual('Sex')","metadata":{"execution":{"iopub.status.busy":"2022-07-27T15:46:47.851359Z","iopub.execute_input":"2022-07-27T15:46:47.851797Z","iopub.status.idle":"2022-07-27T15:46:48.160635Z","shell.execute_reply.started":"2022-07-27T15:46:47.851749Z","shell.execute_reply":"2022-07-27T15:46:48.158801Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Insights:**\n1. The total no. of male passengers is much higher than that of female.\n2. But the survival rate in male(almost 19%) passengers is much lower than that of female(almost 75%).","metadata":{}},{"cell_type":"markdown","source":"**Columns - 'SibSp', 'Parch', 'Family Members'**","metadata":{}},{"cell_type":"code","source":"visual('SibSp')","metadata":{"execution":{"iopub.status.busy":"2022-07-27T15:46:48.162405Z","iopub.execute_input":"2022-07-27T15:46:48.162780Z","iopub.status.idle":"2022-07-27T15:46:48.758063Z","shell.execute_reply.started":"2022-07-27T15:46:48.162747Z","shell.execute_reply":"2022-07-27T15:46:48.757151Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"visual('Parch')","metadata":{"execution":{"iopub.status.busy":"2022-07-27T15:46:48.759568Z","iopub.execute_input":"2022-07-27T15:46:48.761080Z","iopub.status.idle":"2022-07-27T15:46:49.161752Z","shell.execute_reply.started":"2022-07-27T15:46:48.761019Z","shell.execute_reply":"2022-07-27T15:46:49.160357Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.countplot(df['Family Members'])\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-27T15:46:49.163073Z","iopub.execute_input":"2022-07-27T15:46:49.163806Z","iopub.status.idle":"2022-07-27T15:46:49.392900Z","shell.execute_reply.started":"2022-07-27T15:46:49.163751Z","shell.execute_reply":"2022-07-27T15:46:49.391662Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"visual('Family Members')","metadata":{"execution":{"iopub.status.busy":"2022-07-27T15:46:49.394473Z","iopub.execute_input":"2022-07-27T15:46:49.394859Z","iopub.status.idle":"2022-07-27T15:46:49.856166Z","shell.execute_reply.started":"2022-07-27T15:46:49.394823Z","shell.execute_reply":"2022-07-27T15:46:49.854721Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Insights:**\n1. Most of the passengers are travelling alone. No. of people with more than 2 family members are very less.\n2. But the people who are travelling alone has death rate significantly higher than their survival rate.\n3. People with 3 family members has the highest survival rate according to the bar graph.","metadata":{}},{"cell_type":"markdown","source":"**Column - 'Embarked'**","metadata":{}},{"cell_type":"code","source":"sns.countplot(df['Embarked'])\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-27T15:46:49.858180Z","iopub.execute_input":"2022-07-27T15:46:49.858558Z","iopub.status.idle":"2022-07-27T15:46:50.033601Z","shell.execute_reply.started":"2022-07-27T15:46:49.858527Z","shell.execute_reply":"2022-07-27T15:46:50.032067Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"visual('Embarked')","metadata":{"execution":{"iopub.status.busy":"2022-07-27T15:46:50.036207Z","iopub.execute_input":"2022-07-27T15:46:50.036609Z","iopub.status.idle":"2022-07-27T15:46:50.361020Z","shell.execute_reply.started":"2022-07-27T15:46:50.036575Z","shell.execute_reply":"2022-07-27T15:46:50.359688Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Insights:**\n1. Most of the people embarked from Southampton, very less people embarked from Queenstown.\n2. People embarked from 'Southampton' has the lowest survival rate and people embarked from 'Cherbourg' has the highest survival rate.","metadata":{}},{"cell_type":"markdown","source":"**Column - 'Age Category'**","metadata":{}},{"cell_type":"code","source":"sns.countplot(df['Age Category'])\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-27T15:46:50.362982Z","iopub.execute_input":"2022-07-27T15:46:50.363588Z","iopub.status.idle":"2022-07-27T15:46:50.559923Z","shell.execute_reply.started":"2022-07-27T15:46:50.363536Z","shell.execute_reply":"2022-07-27T15:46:50.558566Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"visual('Age Category')","metadata":{"execution":{"iopub.status.busy":"2022-07-27T15:46:50.561589Z","iopub.execute_input":"2022-07-27T15:46:50.562019Z","iopub.status.idle":"2022-07-27T15:46:50.933639Z","shell.execute_reply.started":"2022-07-27T15:46:50.561974Z","shell.execute_reply":"2022-07-27T15:46:50.932085Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Insights:**\n1. Most of the passengers belong from the 20-40 age category(Adults).\n2. Children has the highest survival rate among all.\n3. For adults death rate is much higher than survive rate.","metadata":{}},{"cell_type":"markdown","source":"**Column - 'Fare Category**","metadata":{}},{"cell_type":"code","source":"sns.countplot(df['Fare Category'])\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-27T15:46:50.935386Z","iopub.execute_input":"2022-07-27T15:46:50.935885Z","iopub.status.idle":"2022-07-27T15:46:51.119755Z","shell.execute_reply.started":"2022-07-27T15:46:50.935848Z","shell.execute_reply":"2022-07-27T15:46:51.118566Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"visual('Fare Category')","metadata":{"execution":{"iopub.status.busy":"2022-07-27T15:46:51.121013Z","iopub.execute_input":"2022-07-27T15:46:51.121343Z","iopub.status.idle":"2022-07-27T15:46:51.447376Z","shell.execute_reply.started":"2022-07-27T15:46:51.121314Z","shell.execute_reply":"2022-07-27T15:46:51.445912Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Insights:**\n1. No. of people with low fare tickets are the highest (more than 800 among 891 records). \n2. Low fare people has the lowest rate of survival. High fare people has the highest rate of survival.","metadata":{}},{"cell_type":"markdown","source":"**The target variable - 'Survived'**","metadata":{}},{"cell_type":"code","source":"sns.countplot(df['Survived'])\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-27T15:46:51.449110Z","iopub.execute_input":"2022-07-27T15:46:51.449857Z","iopub.status.idle":"2022-07-27T15:46:51.611264Z","shell.execute_reply.started":"2022-07-27T15:46:51.449804Z","shell.execute_reply":"2022-07-27T15:46:51.610032Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"No. of deceased people is higher than survived people.","metadata":{}},{"cell_type":"markdown","source":"**Taking necessary columns and splitting the data**","metadata":{}},{"cell_type":"code","source":"columns = ['Pclass', 'Sex', 'Age', 'Family Members', 'Embarked', 'Fare Category']","metadata":{"execution":{"iopub.status.busy":"2022-07-27T15:46:51.612606Z","iopub.execute_input":"2022-07-27T15:46:51.612982Z","iopub.status.idle":"2022-07-27T15:46:51.619519Z","shell.execute_reply.started":"2022-07-27T15:46:51.612937Z","shell.execute_reply":"2022-07-27T15:46:51.618194Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x = data[columns]\ny = data['Survived']","metadata":{"execution":{"iopub.status.busy":"2022-07-27T15:46:51.620933Z","iopub.execute_input":"2022-07-27T15:46:51.621333Z","iopub.status.idle":"2022-07-27T15:46:51.632924Z","shell.execute_reply.started":"2022-07-27T15:46:51.621301Z","shell.execute_reply":"2022-07-27T15:46:51.632037Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for col in ['Sex','Embarked', 'Fare Category']:\n    x[col]=LabelEncoder().fit_transform(x[col])","metadata":{"execution":{"iopub.status.busy":"2022-07-27T15:46:51.634172Z","iopub.execute_input":"2022-07-27T15:46:51.635058Z","iopub.status.idle":"2022-07-27T15:46:51.649814Z","shell.execute_reply.started":"2022-07-27T15:46:51.635022Z","shell.execute_reply":"2022-07-27T15:46:51.648322Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_train, x_test, y_train, y_test = train_test_split(x, y, random_state=42, test_size=0.2)","metadata":{"execution":{"iopub.status.busy":"2022-07-27T15:46:51.651710Z","iopub.execute_input":"2022-07-27T15:46:51.652371Z","iopub.status.idle":"2022-07-27T15:46:51.668679Z","shell.execute_reply.started":"2022-07-27T15:46:51.652318Z","shell.execute_reply":"2022-07-27T15:46:51.667238Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Building different models**","metadata":{}},{"cell_type":"code","source":"LR = LogisticRegression().fit(x_train, y_train)\nKNN = KNeighborsClassifier().fit(x_train, y_train)\nDT = DecisionTreeClassifier().fit(x_train, y_train)\nRF = RandomForestClassifier().fit(x_train, y_train)\nNB = GaussianNB().fit(x_train, y_train)\nGBR = GradientBoostingClassifier().fit(x_train, y_train)\nXGB = XGBClassifier().fit(x_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2022-07-27T15:46:51.670409Z","iopub.execute_input":"2022-07-27T15:46:51.670782Z","iopub.status.idle":"2022-07-27T15:46:52.321885Z","shell.execute_reply.started":"2022-07-27T15:46:51.670750Z","shell.execute_reply":"2022-07-27T15:46:52.320869Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mod = [LR,KNN,DT,RF,NB,GBR,XGB]","metadata":{"execution":{"iopub.status.busy":"2022-07-27T15:46:52.323598Z","iopub.execute_input":"2022-07-27T15:46:52.327765Z","iopub.status.idle":"2022-07-27T15:46:52.335574Z","shell.execute_reply.started":"2022-07-27T15:46:52.327706Z","shell.execute_reply":"2022-07-27T15:46:52.334506Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"accuracy = [accuracy_score(y_test, model.predict(x_test)) for model in mod]\nmodels = ['Logistic Regression','K Nearest Neighbour', 'Decision Tree', 'Random Forest',\n         'Naive Bayes', 'Gradient Boosting', 'XgBoost']\nacc = pd.DataFrame({'Models':models,'Accuracy':accuracy})","metadata":{"execution":{"iopub.status.busy":"2022-07-27T15:46:52.338319Z","iopub.execute_input":"2022-07-27T15:46:52.339047Z","iopub.status.idle":"2022-07-27T15:46:52.404123Z","shell.execute_reply.started":"2022-07-27T15:46:52.338994Z","shell.execute_reply":"2022-07-27T15:46:52.400685Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"acc","metadata":{"execution":{"iopub.status.busy":"2022-07-27T15:46:52.407239Z","iopub.execute_input":"2022-07-27T15:46:52.408099Z","iopub.status.idle":"2022-07-27T15:46:52.421299Z","shell.execute_reply.started":"2022-07-27T15:46:52.408046Z","shell.execute_reply":"2022-07-27T15:46:52.419791Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"XgBoost and Gradient Boosting gave the highest accuracy.","metadata":{}},{"cell_type":"code","source":"confusion_matrix(y_test, XGB.predict(x_test))","metadata":{"execution":{"iopub.status.busy":"2022-07-27T15:46:52.423531Z","iopub.execute_input":"2022-07-27T15:46:52.424424Z","iopub.status.idle":"2022-07-27T15:46:52.446665Z","shell.execute_reply.started":"2022-07-27T15:46:52.424372Z","shell.execute_reply":"2022-07-27T15:46:52.445260Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(classification_report(y_test, XGB.predict(x_test)))","metadata":{"execution":{"iopub.status.busy":"2022-07-27T15:46:52.448704Z","iopub.execute_input":"2022-07-27T15:46:52.449263Z","iopub.status.idle":"2022-07-27T15:46:52.470174Z","shell.execute_reply.started":"2022-07-27T15:46:52.449216Z","shell.execute_reply":"2022-07-27T15:46:52.468921Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Testing Data","metadata":{}},{"cell_type":"code","source":"test = pd.read_csv(\"../input/titanic/test.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-07-27T15:46:52.471596Z","iopub.execute_input":"2022-07-27T15:46:52.471997Z","iopub.status.idle":"2022-07-27T15:46:52.484420Z","shell.execute_reply.started":"2022-07-27T15:46:52.471942Z","shell.execute_reply":"2022-07-27T15:46:52.483405Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-27T15:46:52.486368Z","iopub.execute_input":"2022-07-27T15:46:52.487070Z","iopub.status.idle":"2022-07-27T15:46:52.494498Z","shell.execute_reply.started":"2022-07-27T15:46:52.487031Z","shell.execute_reply":"2022-07-27T15:46:52.493407Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.columns","metadata":{"execution":{"iopub.status.busy":"2022-07-27T15:46:52.495987Z","iopub.execute_input":"2022-07-27T15:46:52.496701Z","iopub.status.idle":"2022-07-27T15:46:52.509046Z","shell.execute_reply.started":"2022-07-27T15:46:52.496658Z","shell.execute_reply":"2022-07-27T15:46:52.507613Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ID = test['PassengerId']","metadata":{"execution":{"iopub.status.busy":"2022-07-27T15:46:52.510694Z","iopub.execute_input":"2022-07-27T15:46:52.512104Z","iopub.status.idle":"2022-07-27T15:46:52.521175Z","shell.execute_reply.started":"2022-07-27T15:46:52.512057Z","shell.execute_reply":"2022-07-27T15:46:52.519988Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Checking for nulls in test data**","metadata":{}},{"cell_type":"code","source":"test.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2022-07-27T15:46:52.522492Z","iopub.execute_input":"2022-07-27T15:46:52.523669Z","iopub.status.idle":"2022-07-27T15:46:52.540549Z","shell.execute_reply.started":"2022-07-27T15:46:52.523626Z","shell.execute_reply":"2022-07-27T15:46:52.538993Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#'Fare' and 'Cabin' column won't be present in out dataset so removing the nulls for 'Age' only\ntest[['Age']] = pd.DataFrame(SimpleImputer().fit_transform(test[['Age']]))","metadata":{"execution":{"iopub.status.busy":"2022-07-27T15:46:52.542479Z","iopub.execute_input":"2022-07-27T15:46:52.543499Z","iopub.status.idle":"2022-07-27T15:46:52.555542Z","shell.execute_reply.started":"2022-07-27T15:46:52.543444Z","shell.execute_reply":"2022-07-27T15:46:52.553784Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test['Family Members'] = test['SibSp'] + test['Parch']","metadata":{"execution":{"iopub.status.busy":"2022-07-27T15:46:52.557181Z","iopub.execute_input":"2022-07-27T15:46:52.558315Z","iopub.status.idle":"2022-07-27T15:46:52.569912Z","shell.execute_reply.started":"2022-07-27T15:46:52.558247Z","shell.execute_reply":"2022-07-27T15:46:52.568316Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"t = []\nfor i in range(418):\n    if test['Fare'][i]<100:\n        t.append('Low')\n    elif test['Fare'][i]<250:\n        t.append('Average')\n    else:\n        t.append('High')\ntest['Fare Category'] = t","metadata":{"execution":{"iopub.status.busy":"2022-07-27T15:46:52.572035Z","iopub.execute_input":"2022-07-27T15:46:52.572861Z","iopub.status.idle":"2022-07-27T15:46:52.589846Z","shell.execute_reply.started":"2022-07-27T15:46:52.572806Z","shell.execute_reply":"2022-07-27T15:46:52.588677Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# dropping extra columns\neliminate = [col for col in test.columns if col not in ['Pclass', 'Sex', 'Age', 'Family Members', 'Embarked', 'Fare Category']]\ntest = test.drop(eliminate, axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-07-27T15:46:52.591908Z","iopub.execute_input":"2022-07-27T15:46:52.592809Z","iopub.status.idle":"2022-07-27T15:46:52.607269Z","shell.execute_reply.started":"2022-07-27T15:46:52.592739Z","shell.execute_reply":"2022-07-27T15:46:52.605634Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2022-07-27T15:46:52.609092Z","iopub.execute_input":"2022-07-27T15:46:52.609726Z","iopub.status.idle":"2022-07-27T15:46:52.628385Z","shell.execute_reply.started":"2022-07-27T15:46:52.609686Z","shell.execute_reply":"2022-07-27T15:46:52.627112Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"No more null value present in test data.","metadata":{}},{"cell_type":"code","source":"for col in ['Sex','Embarked', 'Fare Category']:\n    test[col]=LabelEncoder().fit_transform(test[col])","metadata":{"execution":{"iopub.status.busy":"2022-07-27T15:46:52.630698Z","iopub.execute_input":"2022-07-27T15:46:52.631413Z","iopub.status.idle":"2022-07-27T15:46:52.646458Z","shell.execute_reply.started":"2022-07-27T15:46:52.631372Z","shell.execute_reply":"2022-07-27T15:46:52.645388Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred = XGB.predict(test)\nsubmission = pd.DataFrame({'PassengerId':ID, 'Survived':y_pred})","metadata":{"execution":{"iopub.status.busy":"2022-07-27T15:46:52.648072Z","iopub.execute_input":"2022-07-27T15:46:52.649321Z","iopub.status.idle":"2022-07-27T15:46:52.664645Z","shell.execute_reply.started":"2022-07-27T15:46:52.649270Z","shell.execute_reply":"2022-07-27T15:46:52.663416Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-27T15:46:52.665931Z","iopub.execute_input":"2022-07-27T15:46:52.669313Z","iopub.status.idle":"2022-07-27T15:46:52.681036Z","shell.execute_reply.started":"2022-07-27T15:46:52.669255Z","shell.execute_reply":"2022-07-27T15:46:52.680083Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-27T15:46:52.682884Z","iopub.execute_input":"2022-07-27T15:46:52.684880Z","iopub.status.idle":"2022-07-27T15:46:52.697188Z","shell.execute_reply.started":"2022-07-27T15:46:52.684816Z","shell.execute_reply":"2022-07-27T15:46:52.695865Z"},"trusted":true},"execution_count":null,"outputs":[]}]}