{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"<center><img src=\"https://cdn.shopify.com/s/files/1/0616/1606/2711/products/product-image-831446541_1024x1024_d944d4e8-7ca6-4cbc-b7d7-23549e9d6b79_1024x1024.jpg?v=1639755401\" alt=\"drawing\" width=\"300\"/></center>\n\n\n\n","metadata":{}},{"cell_type":"markdown","source":"### I am groot! 👋\n<p style=\"font-size:16px;\">This is a beginner-friendly tutorial that covers a lot of machine learning topics with explanations using Titanic dataset. Hope you will enjoy going through the noutbook. Below you can see what we shall cover, if you find my work helpful, <b>please upvote<b> ⬆️😊.<p>\n    ","metadata":{}},{"cell_type":"markdown","source":"## The menu:\n<p style=\"font-size:16px;\"> <a href=\"#anchor-name\">1. Exploratory data analysis</a><br>\n<a href=\"#anchor-name2\">2. Feature Engineering & Cleaning </a><br>\n<a href=\"#anchor-name3\">3. k-Nearest Neighbors </a><br>\n<a href=\"#anchor-name4\">4. LogisticRegression </a><br>\n<a href=\"#anchor-name5\">5. SVC - Support Vector Classifier </a><br>\n<a href=\"#anchor-name6\">6. Ensemble Learning Methods </a><br>\n<a href=\"#anchor-name7\">7. Summary </a><br>\n</p>\n","metadata":{}},{"cell_type":"markdown","source":"# <a id=\"anchor-name\">Exploratory data analysis:</a>","metadata":{}},{"cell_type":"code","source":"#Here we import required libraries\nimport pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.metrics import roc_curve, accuracy_score\nfrom sklearn.neighbors import KNeighborsClassifier \nfrom sklearn.model_selection import train_test_split, GridSearchCV\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.svm import SVC\nfrom sklearn.tree import DecisionTreeClassifier\nfrom sklearn.ensemble import VotingClassifier\nfrom sklearn.ensemble import BaggingClassifier\nfrom sklearn.ensemble import AdaBoostClassifier\nfrom sklearn.ensemble import RandomForestClassifier\nimport xgboost as xgb","metadata":{"execution":{"iopub.status.busy":"2022-07-11T13:49:29.499208Z","iopub.execute_input":"2022-07-11T13:49:29.499625Z","iopub.status.idle":"2022-07-11T13:49:30.666439Z","shell.execute_reply.started":"2022-07-11T13:49:29.499590Z","shell.execute_reply":"2022-07-11T13:49:30.665313Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Usually, we should always have the idea of what kind of data we are working with, so it is a good habit to go through some rows of data and check out the columns from the beginning. You will have a sense of the data.These are some column definations:\n<p style=\"font-size:14px;\">\nSurvived - Survival Status (0 = No; 1 = Yes) <br>\nPclass   - Passenger Class (1 = 1st; 2 = 2nd; 3 = 3rd)<br>\nSibSp    - Number of Siblings/Spouses Aboard<br>\nParch    - Number of Parents/Children Aboard<br>\nTicket   - Ticket Number<br>\nFare     - Passenger Fare<br>\nCabin    - Cabin<br>\nCmbarked - Port of Embarkation (C = Cherbourg; Q = Queenstown; S = Southampton)<br>\n</p>\n","metadata":{}},{"cell_type":"code","source":"#Load the training data, and print our the first 3 rows\ntrain_df = pd.read_csv('../input/titanic/train.csv')\ntest_df = pd.read_csv('../input/titanic/test.csv')\n\n#save for final evaluation\npassenger_id = test_df['PassengerId']\n\n#Showing the first 3 rows of Training data\ntrain_df.head(3)","metadata":{"execution":{"iopub.status.busy":"2022-07-11T13:49:37.684552Z","iopub.execute_input":"2022-07-11T13:49:37.684938Z","iopub.status.idle":"2022-07-11T13:49:37.733978Z","shell.execute_reply.started":"2022-07-11T13:49:37.684905Z","shell.execute_reply":"2022-07-11T13:49:37.732820Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Total information about the training data\ntrain_df.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-11T13:49:18.403210Z","iopub.status.idle":"2022-07-11T13:49:18.403867Z","shell.execute_reply.started":"2022-07-11T13:49:18.403646Z","shell.execute_reply":"2022-07-11T13:49:18.403668Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Here we find where the missing values are:\ndirty_col_train, dirty_col_test = train_df.columns[train_df.isnull().any()], test_df.columns[test_df.isnull().any()]\nprint(\"Columns missing from Training Data: {}\\nColumns missing from Training Data:{}\".format(dirty_col_train,dirty_col_test))","metadata":{"execution":{"iopub.status.busy":"2022-07-09T19:19:47.64446Z","iopub.execute_input":"2022-07-09T19:19:47.644898Z","iopub.status.idle":"2022-07-09T19:19:47.658033Z","shell.execute_reply.started":"2022-07-09T19:19:47.644862Z","shell.execute_reply":"2022-07-09T19:19:47.656678Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# How much % of the column is missing from Training data\ntrain_df[dirty_col_train].isna().sum().sort_values(ascending=False)/train_df.shape[0] * 100","metadata":{"execution":{"iopub.status.busy":"2022-07-09T19:19:49.684645Z","iopub.execute_input":"2022-07-09T19:19:49.68566Z","iopub.status.idle":"2022-07-09T19:19:49.696712Z","shell.execute_reply.started":"2022-07-09T19:19:49.68562Z","shell.execute_reply":"2022-07-09T19:19:49.695518Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# How much % of the column is missing from Testing data\ntest_df[dirty_col_test].isna().sum().sort_values(ascending=False)/test_df.shape[0] * 100","metadata":{"execution":{"iopub.status.busy":"2022-07-09T19:19:50.924236Z","iopub.execute_input":"2022-07-09T19:19:50.924938Z","iopub.status.idle":"2022-07-09T19:19:50.937111Z","shell.execute_reply.started":"2022-07-09T19:19:50.924886Z","shell.execute_reply":"2022-07-09T19:19:50.935944Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<p style=\"font-size:16px;\"> \nFrom above it is clear that there are missing values in the columns : <b>Cabin </b>, <b>Age </b>, <b>Fare </b> and <b>Embarked </b>. Don't worry, we shall take care of them in feature enginering section. BTW,\nwe can also use seaborn to create a heatmap to see those missing values visually:\n</p>","metadata":{}},{"cell_type":"code","source":"sns.set_style('whitegrid',{'axes.grid' : False})\n#heatmap for missing data of Training set\nplt.figure(figsize=(14,8))\nsns.set(font_scale=1.3)\ng = sns.heatmap(train_df.isnull(),yticklabels=False,cbar=False,cmap='inferno')\ng.set_xticklabels(g.get_xticklabels(), rotation=45, horizontalalignment='right')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-09T19:19:53.066266Z","iopub.execute_input":"2022-07-09T19:19:53.066695Z","iopub.status.idle":"2022-07-09T19:19:53.3429Z","shell.execute_reply.started":"2022-07-09T19:19:53.066661Z","shell.execute_reply":"2022-07-09T19:19:53.34175Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n#Survival analise by differnt metrics\nfig, axs = plt.subplots(ncols=2,nrows=2, figsize=(14,10), sharex=True)\nsns.countplot(x='Survived',data=train_df,palette='inferno',ax=axs[0,0])\nsns.countplot(x='Survived',hue='Sex',data=train_df,palette='inferno',ax=axs[0,1])\nsns.countplot(x='Survived',hue='Pclass',data=train_df,palette='inferno',ax=axs[1,0])\nsns.countplot(x='Survived',hue='Embarked',data=train_df,palette='inferno',ax=axs[1,1])\nfig.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-09T19:19:56.061763Z","iopub.execute_input":"2022-07-09T19:19:56.063093Z","iopub.status.idle":"2022-07-09T19:19:56.743062Z","shell.execute_reply.started":"2022-07-09T19:19:56.063041Z","shell.execute_reply":"2022-07-09T19:19:56.741686Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<p style=\"font-size:16px;\"> \nFirst figure shows us that most people could not survive in the disaster. From other plots, we see that survival rate was way high for women whereas it was the total opposite for men on the board. On top of that, most of those who died were 3rd class passengers. Feel free to share more insights from the graphs below in comments!\n</p>","metadata":{}},{"cell_type":"code","source":"#Histogram of passengers' ages.\nplt.figure(figsize=(10,6))\nsns.histplot(train_df['Age'].dropna(),bins=30, palette='inferno', color='darkred', alpha=0.5, kde=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-09T19:20:03.351171Z","iopub.execute_input":"2022-07-09T19:20:03.35161Z","iopub.status.idle":"2022-07-09T19:20:03.705945Z","shell.execute_reply.started":"2022-07-09T19:20:03.35156Z","shell.execute_reply":"2022-07-09T19:20:03.704761Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<p style=\"font-size:16px;\"> \nWe see that most of passengers were aged somewhere between 20 and 30.\n</p>","metadata":{}},{"cell_type":"code","source":"#Histogram of passengers' fare.\nplt.figure(figsize=(10,6))\nsns.histplot(train_df['Fare'],bins=50, palette='inferno', color='darkred', alpha=0.5)","metadata":{"execution":{"iopub.status.busy":"2022-07-09T19:20:10.145368Z","iopub.execute_input":"2022-07-09T19:20:10.145821Z","iopub.status.idle":"2022-07-09T19:20:10.5042Z","shell.execute_reply.started":"2022-07-09T19:20:10.145785Z","shell.execute_reply":"2022-07-09T19:20:10.503365Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<p style=\"font-size:16px;\"> \nMost people paid less than 50$ for their tickets, suprisingly there are some outliers like the one paid more than 500. Let's see who that person is, and check if he/she survived!\n</p>","metadata":{}},{"cell_type":"code","source":"max_fare = train_df['Fare'].max()\ntrain_df[train_df['Fare']==max_fare]","metadata":{"execution":{"iopub.status.busy":"2022-07-09T19:20:19.955427Z","iopub.execute_input":"2022-07-09T19:20:19.955897Z","iopub.status.idle":"2022-07-09T19:20:19.974927Z","shell.execute_reply.started":"2022-07-09T19:20:19.95586Z","shell.execute_reply":"2022-07-09T19:20:19.973459Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<p style=\"font-size:16px;\">  Umm, they all survived.The most interesting part is the first and last passengers were the servants of  <a href=\"https://titanic.fandom.com/wiki/Thomas_Drake_Martinez_Cardeza\">Thomas Drake Martinez Cardeza</a> who was a wealthy banker. Something is quite off here. Alright, let's move on.<p>","metadata":{}},{"cell_type":"code","source":"#Boxplot of passenger class VS passenger age\nplt.figure(figsize=(10,6))\nsns.boxplot(x='Pclass',y='Age',data=train_df,palette='inferno')","metadata":{"execution":{"iopub.status.busy":"2022-07-09T19:20:25.393217Z","iopub.execute_input":"2022-07-09T19:20:25.394087Z","iopub.status.idle":"2022-07-09T19:20:25.633248Z","shell.execute_reply.started":"2022-07-09T19:20:25.394047Z","shell.execute_reply":"2022-07-09T19:20:25.632045Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<p style=\"font-size:16px;\"> Makes sense, the older the person the more money they tend to have. So, the average  age of 1st class passengers were the highest.<br> You can literally come up with different kind of visulization to undersand the data better, here we stop and start feature engineering.<p>","metadata":{}},{"cell_type":"markdown","source":"# <a id=\"anchor-name2\">Feature Engineering & Cleaning:</a>","metadata":{}},{"cell_type":"markdown","source":"<p style=\"font-size:16px;\">Feature engineering usually all about coming up with more efficient features given the dataset. We will go through columns one by one: PassengerId, Survived, Pclass won't be changed. Regarding the column Name,Sex, Ticket, and Embarked,  we should turn the words/letters into numerical values. We don't do anything to Cabin as we don't use it because of high missing value percentage<p>","metadata":{}},{"cell_type":"code","source":"#Here we shall merge train_df and test_df since the cleaning and feature engineering steps are same.\n# But let's do this trick so we can separate them in the end:\ntest_df['Survived'] = -1\n# Now concatenating Train and Test along rows\ndf = pd.concat([test_df, train_df], axis=0)","metadata":{"execution":{"iopub.status.busy":"2022-07-11T13:49:47.209229Z","iopub.execute_input":"2022-07-11T13:49:47.209658Z","iopub.status.idle":"2022-07-11T13:49:47.228022Z","shell.execute_reply.started":"2022-07-11T13:49:47.209625Z","shell.execute_reply":"2022-07-11T13:49:47.227021Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#We shall introduce new feature: Name_length \ndf['Name_length'] = df['Name'].apply(len)\n\n#We shall introduce new feature: Title\ndf['Title']=df.Name.str.extract('([A-Za-z]+)\\.') \n\n#Seeing all the titles in Training set\ndf['Title'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-07-11T13:49:48.423599Z","iopub.execute_input":"2022-07-11T13:49:48.423972Z","iopub.status.idle":"2022-07-11T13:49:48.443715Z","shell.execute_reply.started":"2022-07-11T13:49:48.423941Z","shell.execute_reply":"2022-07-11T13:49:48.442989Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Minimizing title classes\ndf['Title'] = df['Title'].replace(['Dr', 'Rev','Major', 'Col','Countess', 'Capt', 'Sir', 'Lady', 'Sir', 'Don','Dona', 'Jonkheer'], 'Rare')\ndf['Title'] = df['Title'].replace(['Mlle', 'Ms','Mme'], ['Miss','Miss','Mrs'])\n\n#We can merge SibSp and Parch into one numerical column called Family_size\ndf['Family_size'] = df['SibSp'] + df['Parch'] + 1 #Including the passenger themself\n\ndf['Is_alone'] = df['Family_size'].apply(lambda x: 1 if x == 1 else 0)\n\n#Let's drop the columns we don't use.\ndrop_cols = ['PassengerId', 'Name', 'Ticket', 'Cabin', 'Parch', 'SibSp']\ndf.drop(drop_cols, axis = 1, inplace=True)\n\n#There are some missing values in the Column Age, we shall work on them:\n\n#Let's calculate the mean age according to the Pclass\nage_by_class = df.groupby(['Pclass'])[['Age']].mean().to_dict()['Age']\n\n#The function we use to impute average age to missing values according to the class column:\ndef impute_age(cols):\n    Age = cols[0]\n    Pclass = cols[1]\n    \n    if pd.isnull(Age):\n        return age_by_class[Pclass]\n    else:\n        return Age\n    \n#Imputing the values using apply:\ndf['Age'] = df[['Age','Pclass']].apply(impute_age,axis=1)\n\n#Imputing the values using apply:\ndf['Fare'] = df['Fare'].fillna(df['Fare'].median())\n\n#Remove those rows with no value of 'Embarked'\ndf1 = df.dropna(subset=['Embarked'])\n\n# Let work on categorical values, and turn them into one-hot-encoding , numerical values : Pclass, Sex, Embarked, Title\ndf = pd.get_dummies(df, columns=['Pclass', 'Sex', 'Embarked', 'Title'], drop_first=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-11T13:49:48.747889Z","iopub.execute_input":"2022-07-11T13:49:48.748689Z","iopub.status.idle":"2022-07-11T13:49:48.805822Z","shell.execute_reply.started":"2022-07-11T13:49:48.748657Z","shell.execute_reply":"2022-07-11T13:49:48.804812Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#It is a good practice to standardize  numerical columns:\nscaler = StandardScaler()\ndf[['Age', 'Fare', 'Family_size', 'Name_length']] = scaler.fit_transform(df[['Age', 'Fare', 'Family_size', 'Name_length']])","metadata":{"execution":{"iopub.status.busy":"2022-07-11T13:49:49.446714Z","iopub.execute_input":"2022-07-11T13:49:49.447490Z","iopub.status.idle":"2022-07-11T13:49:49.459099Z","shell.execute_reply.started":"2022-07-11T13:49:49.447443Z","shell.execute_reply":"2022-07-11T13:49:49.458377Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-11T13:49:51.478308Z","iopub.execute_input":"2022-07-11T13:49:51.479000Z","iopub.status.idle":"2022-07-11T13:49:51.495130Z","shell.execute_reply.started":"2022-07-11T13:49:51.478962Z","shell.execute_reply":"2022-07-11T13:49:51.494066Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Now we can separate them just the way we merged before\ntrain_df = df[df[\"Survived\"]!=-1].copy()\ntest_df = df[df[\"Survived\"]==-1].copy()","metadata":{"execution":{"iopub.status.busy":"2022-07-11T13:49:57.845999Z","iopub.execute_input":"2022-07-11T13:49:57.846369Z","iopub.status.idle":"2022-07-11T13:49:57.854190Z","shell.execute_reply.started":"2022-07-11T13:49:57.846338Z","shell.execute_reply":"2022-07-11T13:49:57.853225Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Let's drop the 'Survived' from test set:\ntest_df.drop(\"Survived\", axis = 1, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-11T13:49:59.935588Z","iopub.execute_input":"2022-07-11T13:49:59.935958Z","iopub.status.idle":"2022-07-11T13:49:59.942693Z","shell.execute_reply.started":"2022-07-11T13:49:59.935926Z","shell.execute_reply":"2022-07-11T13:49:59.941980Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# <a id=\"anchor-name3\">k-Nearest Neighbors:</a>","metadata":{}},{"cell_type":"markdown","source":"<p style=\"font-size:16px;\">The k-nearest neighbors (KNN) is a simple supervised machine learning algorithm that can be used to solve both classification and regression problems.</p>\n<center><img src=\"http://res.cloudinary.com/dyd911kmh/image/upload/f_auto,q_auto:best/v1531424125/KNN_final_a1mrv9.png\" alt=\"drawing\" width=\"300\"/></center>\n<p style=\"font-size:16px;\">In simple words, we just look at the N closest neighbors and find out the majority class among them. As you can see, if K=3 new data will be assigned as Class B, but if k=7 it becomes Class A. <br>\n    As k <b>increases</b>, the model's complexity decreases, <b>underfits</b> the data.<br>\nAs k <b>decreases</b>, the model's complecity increases, <b>overfits</b> the data.<br>\n</p>","metadata":{}},{"cell_type":"code","source":"#Let's create dictionary for all model accuracies throughout the notebook:\nmodel_accuracy = dict()\n#Train test split\nX = train_df.drop('Survived', axis = 1)\ny = train_df['Survived']\nX_train, X_test, y_train, y_test = train_test_split(X,y, test_size=0.1, random_state=42, stratify=y)","metadata":{"execution":{"iopub.status.busy":"2022-07-11T13:50:04.284435Z","iopub.execute_input":"2022-07-11T13:50:04.285050Z","iopub.status.idle":"2022-07-11T13:50:04.296827Z","shell.execute_reply.started":"2022-07-11T13:50:04.285019Z","shell.execute_reply":"2022-07-11T13:50:04.296053Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# KNN Classification\n# Let's choose the most optimal k\nneighbors = np.arange(1, 50)\ntrain_acc = {}\ntest_acc = {}\n\nfor neighbor in neighbors:\n    # Set up a KNN Classifier\n    knn = KNeighborsClassifier(n_neighbors=neighbor)\n\n    # Fit the model\n    knn.fit(X_train, y_train)\n  \n    # Compute accuracy\n    train_acc[neighbor] = knn.score(X_train, y_train)\n    test_acc[neighbor] = knn.score(X_test, y_test)\n    \nmodel_accuracy['KNN'] = max(test_acc.values())\nprint(\"The best accuracy: {} when k={}\".format(max(test_acc.values()),max(test_acc, key=test_acc.get)))","metadata":{"execution":{"iopub.status.busy":"2022-07-11T13:50:32.871348Z","iopub.execute_input":"2022-07-11T13:50:32.872289Z","iopub.status.idle":"2022-07-11T13:50:35.458545Z","shell.execute_reply.started":"2022-07-11T13:50:32.872238Z","shell.execute_reply":"2022-07-11T13:50:35.457410Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# We can see the plot here below\nsns.set_style('whitegrid',{'axes.grid' : False})\n# Add a title\nplt.figure(figsize=(10,6))\nplt.title(\"KNN Classification\")\n\n# Plot training accuracies\nplt.plot(neighbors, train_acc.values(), label=\"Training Accuracy\")\n\n# Plot test accuracies\nplt.plot(neighbors, test_acc.values(), label=\"Testing Accuracy\")\n\nplt.legend()\nplt.xlabel(\"Number of Neighbors\")\nplt.ylabel(\"Accuracy\")\n\n# Display the plot\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-11T13:50:38.296392Z","iopub.execute_input":"2022-07-11T13:50:38.296792Z","iopub.status.idle":"2022-07-11T13:50:38.511797Z","shell.execute_reply.started":"2022-07-11T13:50:38.296743Z","shell.execute_reply":"2022-07-11T13:50:38.510783Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<p style=\"font-size:16px;\">The k-nearest neighbors (KNN) performed well, we had the best accuracy when k=27, great , but why stop here. Let's take an adventures through other algorithms in the search of the best accuracy!</p>","metadata":{}},{"cell_type":"markdown","source":"# <a id=\"anchor-name4\">Logistic Regression:</a>\n<p style=\"font-size:16px;\">Logistic regression models the probabilities for classification problems with two possible outcomes. It’s an extension of the linear regression model for classification problems. Try <a href=\"https://www.youtube.com/watch?v=yIYKR4sgzI8&ab_channel=StatQuestwithJoshStarmer\">THIS</a> fun video by StatQuests for more info.</p>","metadata":{"execution":{"iopub.status.busy":"2022-07-09T11:05:11.367056Z","iopub.execute_input":"2022-07-09T11:05:11.367443Z","iopub.status.idle":"2022-07-09T11:05:11.372707Z","shell.execute_reply.started":"2022-07-09T11:05:11.36741Z","shell.execute_reply":"2022-07-09T11:05:11.37147Z"}}},{"cell_type":"code","source":"# Set up a LogisticRegression Classifier\nlr = LogisticRegression(solver='lbfgs', max_iter=1000)\n\n# Fit the model\nlr.fit(X_train, y_train)\n\n# Compute accuracy\nprint(lr.score(X_test, y_test))\nprint(lr.score(X_train, y_train))\n","metadata":{"execution":{"iopub.status.busy":"2022-07-11T13:50:48.938610Z","iopub.execute_input":"2022-07-11T13:50:48.938978Z","iopub.status.idle":"2022-07-11T13:50:48.991418Z","shell.execute_reply.started":"2022-07-11T13:50:48.938948Z","shell.execute_reply":"2022-07-11T13:50:48.990134Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<p style=\"font-size:16px;\">Let's try playing with its hyperparameters, seems like we can squeeze some more out of this underfitting attaboy. Unlike the case in KNN, we shall use better method called 'GridSearchCV'. It just tries different values of parameters and find the best estimor for us, it can also perform CrossValidation at the same time. Wonderful, huh!</p>","metadata":{}},{"cell_type":"code","source":"# Set up a LogisticRegression Classifier:\nlinear_classifier = LogisticRegression(solver='lbfgs', max_iter=1000)\n\n# Showing the parameters and initiliazing the GridSearch5-FoldCv:\nparameters = {'C':[1, 10, 100, 1000, 10000]}\nsearcher = GridSearchCV(linear_classifier, parameters, cv=5)\nsearcher.fit(X_train, y_train)\n\nbest_acc = searcher.score(X_test, y_test)\n#Adding the best score to the final dict\nmodel_accuracy['Logistic Regression'] = searcher.score(X_test, y_test)\n\n# Reporting the best values:\nprint(\"The best accuracy: {} when C={}\".format(best_acc,searcher.best_params_['C']))\n\n#list of prediction probabilities for ROC curve\ny_pred_probs = searcher.predict_proba(X_test)[:, 1]","metadata":{"execution":{"iopub.status.busy":"2022-07-11T13:51:04.299283Z","iopub.execute_input":"2022-07-11T13:51:04.300464Z","iopub.status.idle":"2022-07-11T13:51:04.895303Z","shell.execute_reply.started":"2022-07-11T13:51:04.300420Z","shell.execute_reply":"2022-07-11T13:51:04.894065Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<p style=\"font-size:20px;\">Voila! We have <b>85%</b> accuracy after tweaking the model.</p>","metadata":{}},{"cell_type":"code","source":"# The ROC Curve is mostly used along confusion matrix when working with classification problems\n# The closer the figure to the left corner , or the more the areas under the curver, the better it is:\nplt.figure(figsize=(10,6))\n\n# Generate ROC curve values: fpr, tpr, thresholds\nfpr, tpr, thresholds = roc_curve(y_test, y_pred_probs)\n\nplt.plot([0, 1], [0, 1], 'k--')\n\nplt.plot([0, 0], [0, 1], 'r--')\nplt.plot([1, 0], [1, 1], 'r--')\n\n# Plot tpr against fpr\nplt.plot(fpr, tpr)\nplt.xlabel('False Positive Rate')\nplt.ylabel('True Positive Rate')\nplt.title('ROC')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-11T13:51:14.311306Z","iopub.execute_input":"2022-07-11T13:51:14.311665Z","iopub.status.idle":"2022-07-11T13:51:14.549475Z","shell.execute_reply.started":"2022-07-11T13:51:14.311637Z","shell.execute_reply":"2022-07-11T13:51:14.548429Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# <a id=\"anchor-name5\">SVC - Support Vector Classifier</a>\n<p style=\"font-size:16px;\">Support Vector Machine (SVM) is a supervised machine learning algorithm that can be used for both classification or regression problems.A support vector machine (SVM) is a supervised machine learning algorithm that solves two-group classification problems. <a href=\"https://www.youtube.com/watch?v=efR1C6CvhmE&ab_channel=StatQuestwithJoshStarmer\">THIS</a> musical video by StatQuests comes to rescue for more info.</p>","metadata":{}},{"cell_type":"code","source":"#Since we got really goot at training data, let's start right from here:\n# Instantiate an SVM\nsvm = SVC()\n\n# Showing the parameters and initiliazing the GridSearch5-FoldCv:\nparameters = {'C':[0.1, 1, 10, 20,30], 'gamma':[0.01, 0.03, 0.1, 0.3, 1]}\nsearcher = GridSearchCV(svm, parameters, cv=5)\nsearcher.fit(X_train, y_train)\n\nbest_acc = searcher.score(X_test, y_test)\n#Adding the best score to the final dict\nmodel_accuracy['SVC'] = searcher.score(X_test, y_test)\n\n# Reporting the best values:\nprint(\"The best accuracy: {} when {}\".format(best_acc,searcher.best_params_))\n","metadata":{"execution":{"iopub.status.busy":"2022-07-11T13:51:18.604198Z","iopub.execute_input":"2022-07-11T13:51:18.604607Z","iopub.status.idle":"2022-07-11T13:51:22.208814Z","shell.execute_reply.started":"2022-07-11T13:51:18.604575Z","shell.execute_reply":"2022-07-11T13:51:22.207691Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<p style=\"font-size:20px;\">Voila! We have <b>80%</b> accuracy, and SVC sucked! Anyway, now let's start ensemble methods!</p>","metadata":{"execution":{"iopub.status.busy":"2022-07-09T13:19:36.724133Z","iopub.execute_input":"2022-07-09T13:19:36.72486Z","iopub.status.idle":"2022-07-09T13:19:36.732342Z","shell.execute_reply.started":"2022-07-09T13:19:36.724808Z","shell.execute_reply":"2022-07-09T13:19:36.730789Z"}}},{"cell_type":"markdown","source":"# <a id=\"anchor-name6\">Ensemble Methods</a>\n<p style=\"font-size:16px;\">There are many ensemble learning methods for classification problems.Ensemble learning is a general meta approach to machine learning that seeks better predictive performance by combining the predictions from different models put together.</p>\n\n## Classification and Regression Tree (CART):\n","metadata":{}},{"cell_type":"code","source":"# Instantiate the DecisionTreeClassifier\ndt = DecisionTreeClassifier()\n\n# Showing the parameters and initiliazing the GridSearch5-FoldCv:\nparameters = {'max_depth':list(range(1,15))}\nsearcher = GridSearchCV(dt, parameters, cv=5)\nsearcher.fit(X_train, y_train)\n\nbest_acc = searcher.score(X_test, y_test)\n#Adding the best score to the final dict\nmodel_accuracy['Decision Tree'] = searcher.score(X_test, y_test)\n\n# Reporting the best values:\nprint(\"The best accuracy: {} when {}\".format(best_acc,searcher.best_params_))","metadata":{"execution":{"iopub.status.busy":"2022-07-11T13:51:27.693247Z","iopub.execute_input":"2022-07-11T13:51:27.693658Z","iopub.status.idle":"2022-07-11T13:51:28.087073Z","shell.execute_reply.started":"2022-07-11T13:51:27.693622Z","shell.execute_reply":"2022-07-11T13:51:28.085994Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## <a id=\"anchor-name5\">Voting Classifier:</a>","metadata":{}},{"cell_type":"code","source":"# Instantiate lr\nlr = LogisticRegression()\n\n# Instantiate knn\nknn = KNeighborsClassifier(n_neighbors=42)\n\n# Instantiate dt\ndt = DecisionTreeClassifier(max_depth=4)\n\n# List of Classifiers\nclassifiers = [('Logistic Regression', lr), ('K Nearest Neighbours', knn), ('Classification Tree', dt)]\n\n# Instantiate a VotingClassifier\nvc = VotingClassifier(estimators=classifiers, voting='soft')\n# Fit vc to the training set\nvc.fit(X_train, y_train)   \n\ny_pred = vc.predict(X_test)\n\n# Calculate accuracy\naccuracy = accuracy_score(y_test, y_pred)\nmodel_accuracy['Voting Classifier'] = accuracy\nprint('The accuracy: {:.2f}'.format(accuracy))","metadata":{"execution":{"iopub.status.busy":"2022-07-11T13:51:33.130123Z","iopub.execute_input":"2022-07-11T13:51:33.130471Z","iopub.status.idle":"2022-07-11T13:51:33.195440Z","shell.execute_reply.started":"2022-07-11T13:51:33.130444Z","shell.execute_reply":"2022-07-11T13:51:33.194275Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Random Forest Classifier","metadata":{}},{"cell_type":"code","source":"# Instantiate the RandomForestClassifier\nrf = RandomForestClassifier()\n\n# Showing the parameters and initiliazing the GridSearch5-FoldCv:\nparams_rf = {'n_estimators':[100,200,300,400,500],\n            'max_features':['log2','auto','sqrt'],\n            'min_samples_leaf':[1,2,4,16,32,64]}\n\ngrid_rf = GridSearchCV(estimator=rf,\n                       param_grid=params_rf,\n                       cv=5,\n                       n_jobs=-1)\n\ngrid_rf.fit(X_train, y_train)\n\nbest_acc = grid_rf.score(X_test, y_test)\n\n#Adding the best score to the final dict\nmodel_accuracy['Random Forest Classifier'] = grid_rf.score(X_test, y_test)\n\n# Reporting the best values:\nprint(\"The best accuracy: {} when {}\".format(best_acc,grid_rf.best_params_))\n\n#=-=-=-=-=-=-=-=-=-=-=-=-=-=-=-=-=-=-=-=-=-=-=\n#  Added later\nmodel_final_eval =  grid_rf.best_estimator_","metadata":{"execution":{"iopub.status.busy":"2022-07-11T13:51:36.011057Z","iopub.execute_input":"2022-07-11T13:51:36.011787Z","iopub.status.idle":"2022-07-11T13:53:23.987762Z","shell.execute_reply.started":"2022-07-11T13:51:36.011736Z","shell.execute_reply":"2022-07-11T13:53:23.986283Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## <a id=\"anchor-name5\">Bagging Classifier:</a>","metadata":{}},{"cell_type":"code","source":"# Instantiate DecisionTreeClassifier\ndt = DecisionTreeClassifier(random_state=1,max_depth=3)\n\n# Instantiate BaggingClassifier\nbc = BaggingClassifier(base_estimator=dt, n_estimators=30, random_state=1)\n\n# Fitting\nbc.fit(X_train, y_train)\n\n# Predicting\ny_pred = bc.predict(X_test)\n\n# Evaluate acc_test\nacc_test = accuracy_score(y_test, y_pred)\nmodel_accuracy['Bagging Classifier'] = acc_test\nprint('Test set accuracy: {}'.format(acc_test)) ","metadata":{"execution":{"iopub.status.busy":"2022-07-11T13:53:46.277238Z","iopub.execute_input":"2022-07-11T13:53:46.277620Z","iopub.status.idle":"2022-07-11T13:53:46.368098Z","shell.execute_reply.started":"2022-07-11T13:53:46.277589Z","shell.execute_reply":"2022-07-11T13:53:46.367011Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## <a id=\"anchor-name5\">AdaBoost Classifier:</a>","metadata":{}},{"cell_type":"code","source":"# Instantiate DecisionTreeClassifier\ndt = DecisionTreeClassifier(max_depth=1)\n\n# Instantiate AdaBoostClassifier\nada = AdaBoostClassifier(base_estimator=dt, n_estimators=20, random_state=42)\n                      \n# Fitting\nada.fit(X_train, y_train)\n                         \n# Predicting\ny_pred = ada.predict(X_test)\n\n# Evaluate acc_test\nacc_test = accuracy_score(y_test, y_pred)\nmodel_accuracy['AdaBoost Classifier'] = acc_test\nprint('Test set accuracy: {}'.format(acc_test)) ","metadata":{"execution":{"iopub.status.busy":"2022-07-11T13:54:17.645379Z","iopub.execute_input":"2022-07-11T13:54:17.645815Z","iopub.status.idle":"2022-07-11T13:54:17.702780Z","shell.execute_reply.started":"2022-07-11T13:54:17.645778Z","shell.execute_reply":"2022-07-11T13:54:17.701972Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## XGBoost Classifier","metadata":{}},{"cell_type":"code","source":"# Showing the parameters and initiliazing the GridSearch5-FoldCv:\ngb_param_grid = {\n    'n_estimators': [10,20,30,40],\n    'max_depth': list(range(1,11)),\n    'objective':['binary:logistic']\n}\n\n# Instantiating the XGBClassifier\ngb = xgb.XGBClassifier(seed=123)\n\ngrid = GridSearchCV(estimator=gb, param_grid=gb_param_grid, cv=5)\n\ngrid.fit(X_train, y_train)\n\nbest_acc = grid.score(X_test, y_test)\n#Adding the best score to the final dict\nmodel_accuracy['XGBoost Classifier'] = best_acc\n\n# Reporting the best values:\nprint(\"The best accuracy: {} when {}\".format(best_acc,grid.best_params_))","metadata":{"execution":{"iopub.status.busy":"2022-07-11T13:54:19.545135Z","iopub.execute_input":"2022-07-11T13:54:19.545870Z","iopub.status.idle":"2022-07-11T13:54:38.606099Z","shell.execute_reply.started":"2022-07-11T13:54:19.545830Z","shell.execute_reply":"2022-07-11T13:54:38.605042Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<p style=\"font-size:16px;\">All ensembel method performed well. I beleive some of them can still be improved by tweaking the hyperparameters. Among them, XGBoost did a pretty good  job.</p>","metadata":{}},{"cell_type":"markdown","source":"# <a id=\"anchor-name7\">Summary</a>","metadata":{}},{"cell_type":"markdown","source":"<p style=\"font-size:18px;\">In this notebook, we took the raw Titanic dataset ,and went through different stages ranging from data exploration to model evaluation. We trained 9 models, some are quite similar to each other, namely ensemble learning algorithms. The best score obtained was 85% by LogisticRegression, following is XGBoost with 84% accuracy. I really hope you will enjoy this notebook just the way I did. I would love to know your opinions and criticism regarding the notebook. After all, please do not forget to UpVote the notebook, if you liked my work.</p>","metadata":{"execution":{"iopub.status.busy":"2022-07-09T16:48:57.067899Z","iopub.execute_input":"2022-07-09T16:48:57.068389Z","iopub.status.idle":"2022-07-09T16:48:57.075681Z","shell.execute_reply.started":"2022-07-09T16:48:57.068356Z","shell.execute_reply":"2022-07-09T16:48:57.074316Z"}}},{"cell_type":"code","source":"import plotly.graph_objs as go\nimport plotly.offline as py\ny = list(model_accuracy.values())\nx = list(model_accuracy.keys())\ndata = [go.Bar(\n            x= x,\n            y= y,\n            width = 0.5,\n            marker=dict(\n            color = list(model_accuracy.values()),\n            colorscale='viridis',\n            showscale=True,\n            reversescale = False\n            ),\n            opacity=0.5\n        )]\n\nlayout= go.Layout(\n    autosize= True,\n    title= 'Barplot of Different Classification Algorithms',\n    yaxis=dict(\n        title= 'Accuracy',\n        ticklen= 5,\n        gridwidth= 2\n    )\n)\nfig = go.Figure(data=data, layout=layout)\nfig.update_layout(barmode='stack', xaxis={'categoryorder':'total descending'},title_x=0.5)\npy.iplot(fig, filename='bar-acc')","metadata":{"execution":{"iopub.status.busy":"2022-07-10T11:06:08.654765Z","iopub.execute_input":"2022-07-10T11:06:08.655158Z","iopub.status.idle":"2022-07-10T11:06:08.701311Z","shell.execute_reply.started":"2022-07-10T11:06:08.655125Z","shell.execute_reply":"2022-07-10T11:06:08.700258Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Final submission\npredictions = grid.predict(test_df)\nSubmission = pd.DataFrame({ 'PassengerId': passenger_id,\n                            'Survived': predictions })\nSubmission.to_csv(\"FinalSubmission.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-11T13:55:08.628980Z","iopub.execute_input":"2022-07-11T13:55:08.629522Z","iopub.status.idle":"2022-07-11T13:55:08.656649Z","shell.execute_reply.started":"2022-07-11T13:55:08.629479Z","shell.execute_reply":"2022-07-11T13:55:08.655338Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}