{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## 1. Import the required libraries and modules:","metadata":{}},{"cell_type":"code","source":"# Import the relevant libraries\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport pandas as pd\nimport collections\nimport seaborn as sns\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.neighbors import KNeighborsClassifier\nfrom sklearn.tree import DecisionTreeClassifier\nfrom sklearn.ensemble import RandomForestClassifier\nimport warnings\nwarnings.filterwarnings('ignore')","metadata":{"execution":{"iopub.status.busy":"2022-07-04T13:58:34.367529Z","iopub.execute_input":"2022-07-04T13:58:34.368204Z","iopub.status.idle":"2022-07-04T13:58:34.375037Z","shell.execute_reply.started":"2022-07-04T13:58:34.368165Z","shell.execute_reply":"2022-07-04T13:58:34.373734Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 2. Read training & test dataset","metadata":{}},{"cell_type":"code","source":"# Read training & test dataset\ndf_train = pd.read_csv('../input/titanic/train.csv')\ndf_test = pd.read_csv(\"../input/titanic/test.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-07-04T13:31:11.249979Z","iopub.execute_input":"2022-07-04T13:31:11.250466Z","iopub.status.idle":"2022-07-04T13:31:11.269956Z","shell.execute_reply.started":"2022-07-04T13:31:11.250425Z","shell.execute_reply":"2022-07-04T13:31:11.268805Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 3. EDA","metadata":{}},{"cell_type":"code","source":"# Display training set head\ndf_train.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-04T13:31:14.744019Z","iopub.execute_input":"2022-07-04T13:31:14.745596Z","iopub.status.idle":"2022-07-04T13:31:14.764516Z","shell.execute_reply.started":"2022-07-04T13:31:14.745515Z","shell.execute_reply":"2022-07-04T13:31:14.763189Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Display test set head\ndf_test.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-04T13:31:21.016913Z","iopub.execute_input":"2022-07-04T13:31:21.017339Z","iopub.status.idle":"2022-07-04T13:31:21.034912Z","shell.execute_reply.started":"2022-07-04T13:31:21.017305Z","shell.execute_reply":"2022-07-04T13:31:21.033724Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Method to drip non-numeric variables\ndef drop_culumns(df):\n    df = df.drop(['Name','Ticket','Cabin'], axis=1)\n    return df\n\ndef encode_embarked(df):\n    df['Embarked'] = df['Embarked'].map({'C':0, 'Q':1, 'S':2})\n    return df\n\n# Method to encode sex variable\ndef encode_sex(df):\n    df['Sex'] = df['Sex'].map({'male': 0, 'female':1})\n    return df","metadata":{"execution":{"iopub.status.busy":"2022-07-04T13:30:13.365767Z","iopub.execute_input":"2022-07-04T13:30:13.366196Z","iopub.status.idle":"2022-07-04T13:30:13.374567Z","shell.execute_reply.started":"2022-07-04T13:30:13.36616Z","shell.execute_reply":"2022-07-04T13:30:13.372951Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Drop non-numeric variables\ndf_train = drop_culumns(df_train)\ndf_test = drop_culumns(df_test)\n\n# Encode Sex variables\ndf_train = encode_sex(df_train)\ndf_test = encode_sex(df_test)\n\n# Encode Embarked variables \ndf_train = encode_embarked(df_train)\ndf_test = encode_embarked(df_test)","metadata":{"execution":{"iopub.status.busy":"2022-07-04T13:31:37.40571Z","iopub.execute_input":"2022-07-04T13:31:37.406166Z","iopub.status.idle":"2022-07-04T13:31:37.427747Z","shell.execute_reply.started":"2022-07-04T13:31:37.406131Z","shell.execute_reply":"2022-07-04T13:31:37.426718Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Display the new training dataset\ndf_train.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-04T13:31:55.63423Z","iopub.execute_input":"2022-07-04T13:31:55.634678Z","iopub.status.idle":"2022-07-04T13:31:55.650841Z","shell.execute_reply.started":"2022-07-04T13:31:55.634642Z","shell.execute_reply":"2022-07-04T13:31:55.650032Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Display the new test dataset\ndf_test.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-04T13:32:15.679542Z","iopub.execute_input":"2022-07-04T13:32:15.679982Z","iopub.status.idle":"2022-07-04T13:32:15.69431Z","shell.execute_reply.started":"2022-07-04T13:32:15.679946Z","shell.execute_reply":"2022-07-04T13:32:15.692832Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Data Explotory Analysis (Training set)\ndf_train.describe()","metadata":{"execution":{"iopub.status.busy":"2022-07-04T13:32:31.147846Z","iopub.execute_input":"2022-07-04T13:32:31.148261Z","iopub.status.idle":"2022-07-04T13:32:31.195337Z","shell.execute_reply.started":"2022-07-04T13:32:31.148228Z","shell.execute_reply":"2022-07-04T13:32:31.194475Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plot a survival pie chart for training set\nplt.pie(df_train['Survived'].value_counts(), labels = ['Not Survived', 'Survived'], explode = (0.1,0.1), shadow = True,autopct='%1.1f%%')\nplt.title(\"Titanic Survival Pie Chart for training set\", bbox={'facecolor':'0.8', 'pad':5})\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-04T13:32:47.78682Z","iopub.execute_input":"2022-07-04T13:32:47.787223Z","iopub.status.idle":"2022-07-04T13:32:47.948895Z","shell.execute_reply.started":"2022-07-04T13:32:47.787181Z","shell.execute_reply":"2022-07-04T13:32:47.947255Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plot a gender pie chart for training set\nplt.pie(df_train['Sex'].value_counts(), labels = ['Male', 'Female'], explode = (0.1,0.1), shadow = True,autopct='%1.1f%%')\nplt.title(\"Titanic Gender Pie Chart for Training set\", bbox={'facecolor':'0.8', 'pad':5})\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-04T13:33:05.452424Z","iopub.execute_input":"2022-07-04T13:33:05.452858Z","iopub.status.idle":"2022-07-04T13:33:05.570918Z","shell.execute_reply.started":"2022-07-04T13:33:05.452822Z","shell.execute_reply":"2022-07-04T13:33:05.569574Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plot box rating by users\nsns.set_theme(style=\"whitegrid\")\nax = sns.boxplot(x= df_train['Age'])","metadata":{"execution":{"iopub.status.busy":"2022-07-04T13:33:23.060317Z","iopub.execute_input":"2022-07-04T13:33:23.060765Z","iopub.status.idle":"2022-07-04T13:33:23.290483Z","shell.execute_reply.started":"2022-07-04T13:33:23.060727Z","shell.execute_reply":"2022-07-04T13:33:23.289353Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train['Age'].describe()","metadata":{"execution":{"iopub.status.busy":"2022-07-04T13:33:40.961712Z","iopub.execute_input":"2022-07-04T13:33:40.962116Z","iopub.status.idle":"2022-07-04T13:33:40.976179Z","shell.execute_reply.started":"2022-07-04T13:33:40.962083Z","shell.execute_reply":"2022-07-04T13:33:40.974916Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plot bar graph of Ticket class for training set\nplt.figure(figsize=(10, 5))\nsns.countplot(df_train['Pclass'])\nplt.title('Ticket Class for Training set')\nplt.xlabel('Pclass')\nplt.ylabel('No. of passengers in each class')","metadata":{"execution":{"iopub.status.busy":"2022-07-04T13:33:58.371185Z","iopub.execute_input":"2022-07-04T13:33:58.372206Z","iopub.status.idle":"2022-07-04T13:33:58.579863Z","shell.execute_reply.started":"2022-07-04T13:33:58.372149Z","shell.execute_reply":"2022-07-04T13:33:58.579023Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Data Explotory Analysis (Test set)\ndf_test.describe()","metadata":{"execution":{"iopub.status.busy":"2022-07-04T13:34:22.576562Z","iopub.execute_input":"2022-07-04T13:34:22.57703Z","iopub.status.idle":"2022-07-04T13:34:22.616167Z","shell.execute_reply.started":"2022-07-04T13:34:22.576994Z","shell.execute_reply":"2022-07-04T13:34:22.614853Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plot a gender pie chart for test set\nplt.pie(df_test['Sex'].value_counts(), labels = ['Male', 'Female'], explode = (0.1,0.1), shadow = True,autopct='%1.1f%%')\nplt.title(\"Titanic Gender Pie Chart for test set\", bbox={'facecolor':'0.8', 'pad':5})\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-04T13:34:36.474555Z","iopub.execute_input":"2022-07-04T13:34:36.475043Z","iopub.status.idle":"2022-07-04T13:34:36.598103Z","shell.execute_reply.started":"2022-07-04T13:34:36.475005Z","shell.execute_reply":"2022-07-04T13:34:36.59658Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plot box rating by users\nsns.set_theme(style=\"whitegrid\")\nax = sns.boxplot(x= df_test['Age'])","metadata":{"execution":{"iopub.status.busy":"2022-07-04T13:34:54.585712Z","iopub.execute_input":"2022-07-04T13:34:54.586158Z","iopub.status.idle":"2022-07-04T13:34:54.75869Z","shell.execute_reply.started":"2022-07-04T13:34:54.586123Z","shell.execute_reply":"2022-07-04T13:34:54.757448Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plot bar graph of Ticket class for test set\nplt.figure(figsize=(10, 5))\nsns.countplot(df_test['Pclass'])\nplt.title('Ticket Class for Test set')\nplt.xlabel('Pclass')\nplt.ylabel('No. of passengers in each class')","metadata":{"execution":{"iopub.status.busy":"2022-07-04T13:35:09.68552Z","iopub.execute_input":"2022-07-04T13:35:09.685953Z","iopub.status.idle":"2022-07-04T13:35:09.840346Z","shell.execute_reply.started":"2022-07-04T13:35:09.685921Z","shell.execute_reply":"2022-07-04T13:35:09.838816Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Split data into x and y\nY = df_train['Survived']\nX = df_train.drop(labels='Survived', axis = 1)","metadata":{"execution":{"iopub.status.busy":"2022-07-04T13:41:52.11956Z","iopub.execute_input":"2022-07-04T13:41:52.120018Z","iopub.status.idle":"2022-07-04T13:41:52.127441Z","shell.execute_reply.started":"2022-07-04T13:41:52.119976Z","shell.execute_reply":"2022-07-04T13:41:52.126175Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Filling NAN values with median \nX.loc[X['Age'].isna(), 'Age'] = X['Age'].median()\nX.loc[X['Embarked'].isna(), 'Embarked'] = X['Embarked'].median()","metadata":{"execution":{"iopub.status.busy":"2022-07-04T13:43:07.684756Z","iopub.execute_input":"2022-07-04T13:43:07.686018Z","iopub.status.idle":"2022-07-04T13:43:07.696307Z","shell.execute_reply.started":"2022-07-04T13:43:07.685961Z","shell.execute_reply":"2022-07-04T13:43:07.695352Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 4. Split the data into traininng and test sets","metadata":{}},{"cell_type":"code","source":"x_train, x_test, y_train, y_test = train_test_split(X, Y, test_size = 0.2, random_state = 42)","metadata":{"execution":{"iopub.status.busy":"2022-07-04T13:50:14.476641Z","iopub.execute_input":"2022-07-04T13:50:14.478139Z","iopub.status.idle":"2022-07-04T13:50:14.486535Z","shell.execute_reply.started":"2022-07-04T13:50:14.478055Z","shell.execute_reply":"2022-07-04T13:50:14.485542Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 5. Train the models","metadata":{}},{"cell_type":"markdown","source":"## 5.1. Logistic regression model","metadata":{}},{"cell_type":"code","source":"# Implementing the logistic regression \nmodelLogistic = LogisticRegression()\nmodelLogistic.fit(x_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2022-07-04T13:53:01.940591Z","iopub.execute_input":"2022-07-04T13:53:01.941292Z","iopub.status.idle":"2022-07-04T13:53:01.981573Z","shell.execute_reply.started":"2022-07-04T13:53:01.941251Z","shell.execute_reply":"2022-07-04T13:53:01.980715Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Accuracy using Logistic regression \", modelLogistic.score(x_test, y_test)*100)","metadata":{"execution":{"iopub.status.busy":"2022-07-04T13:51:58.003457Z","iopub.execute_input":"2022-07-04T13:51:58.004299Z","iopub.status.idle":"2022-07-04T13:51:58.01357Z","shell.execute_reply.started":"2022-07-04T13:51:58.004254Z","shell.execute_reply":"2022-07-04T13:51:58.012477Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 5.2. K Nearest Neighbors","metadata":{}},{"cell_type":"code","source":"# Train using KNN\nfor i in range(1,20):\n    knn = KNeighborsClassifier(n_neighbors=i)\n    knn.fit(x_train, y_train)\n    print(\"No of neighbors -\", i, \", Accuracy using K nearest neighbors\" ,knn.score(x_test,y_test) * 100)","metadata":{"execution":{"iopub.status.busy":"2022-07-04T13:55:18.667154Z","iopub.execute_input":"2022-07-04T13:55:18.667576Z","iopub.status.idle":"2022-07-04T13:55:18.893651Z","shell.execute_reply.started":"2022-07-04T13:55:18.667542Z","shell.execute_reply":"2022-07-04T13:55:18.892246Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 5.3. Decision Trees","metadata":{}},{"cell_type":"code","source":"# Train using decision trees\ndt = DecisionTreeClassifier()\ndt.fit(x_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2022-07-04T13:56:38.84489Z","iopub.execute_input":"2022-07-04T13:56:38.845325Z","iopub.status.idle":"2022-07-04T13:56:38.860188Z","shell.execute_reply.started":"2022-07-04T13:56:38.845288Z","shell.execute_reply":"2022-07-04T13:56:38.859378Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Accuracy using decision trees \" ,dt.score(x_test,y_test) * 100)","metadata":{"execution":{"iopub.status.busy":"2022-07-04T13:57:00.917973Z","iopub.execute_input":"2022-07-04T13:57:00.918778Z","iopub.status.idle":"2022-07-04T13:57:00.927351Z","shell.execute_reply.started":"2022-07-04T13:57:00.918738Z","shell.execute_reply":"2022-07-04T13:57:00.926368Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 5.4. Random Forest Trees","metadata":{}},{"cell_type":"code","source":"# Train using random forest trees\nrf = RandomForestClassifier()\nrf.fit(x_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2022-07-04T13:58:43.885124Z","iopub.execute_input":"2022-07-04T13:58:43.885563Z","iopub.status.idle":"2022-07-04T13:58:44.122093Z","shell.execute_reply.started":"2022-07-04T13:58:43.885528Z","shell.execute_reply":"2022-07-04T13:58:44.121014Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Accuracy using random forest trees \" ,rf.score(x_test,y_test) * 100)","metadata":{"execution":{"iopub.status.busy":"2022-07-04T13:58:59.580931Z","iopub.execute_input":"2022-07-04T13:58:59.581804Z","iopub.status.idle":"2022-07-04T13:58:59.610224Z","shell.execute_reply.started":"2022-07-04T13:58:59.581745Z","shell.execute_reply":"2022-07-04T13:58:59.609132Z"},"trusted":true},"execution_count":null,"outputs":[]}]}