{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"#data analysis libraries \nimport numpy as np\nimport pandas as pd\n\n#visualization libraries\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n%matplotlib inline\n\n#ignore warnings\nimport warnings\nwarnings.filterwarnings('ignore')","metadata":{"execution":{"iopub.status.busy":"2022-08-11T14:06:53.068841Z","iopub.execute_input":"2022-08-11T14:06:53.069297Z","iopub.status.idle":"2022-08-11T14:06:53.077441Z","shell.execute_reply.started":"2022-08-11T14:06:53.069255Z","shell.execute_reply":"2022-08-11T14:06:53.076287Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#import titanic dataset\n#import train and test CSV files\ntrain = pd.read_csv(\"../input/titanic/train.csv\")\ntest = pd.read_csv(\"../input/titanic/test.csv\")\n\n#take a look at the training data\ntrain.describe(include=\"all\")","metadata":{"execution":{"iopub.status.busy":"2022-08-11T14:06:53.134477Z","iopub.execute_input":"2022-08-11T14:06:53.135236Z","iopub.status.idle":"2022-08-11T14:06:53.189113Z","shell.execute_reply.started":"2022-08-11T14:06:53.135190Z","shell.execute_reply":"2022-08-11T14:06:53.188286Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#get a list of the features within the dataset\nprint(train.columns)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T14:06:53.193804Z","iopub.execute_input":"2022-08-11T14:06:53.194173Z","iopub.status.idle":"2022-08-11T14:06:53.200193Z","shell.execute_reply.started":"2022-08-11T14:06:53.194139Z","shell.execute_reply":"2022-08-11T14:06:53.199075Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#see a sample of the dataset to get an idea of the variables\ntrain.sample(5)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T14:06:53.259179Z","iopub.execute_input":"2022-08-11T14:06:53.259657Z","iopub.status.idle":"2022-08-11T14:06:53.277477Z","shell.execute_reply.started":"2022-08-11T14:06:53.259617Z","shell.execute_reply":"2022-08-11T14:06:53.276695Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#check for any other unusable values\nprint(pd.isnull(train).sum())","metadata":{"execution":{"iopub.status.busy":"2022-08-11T14:06:53.322004Z","iopub.execute_input":"2022-08-11T14:06:53.323027Z","iopub.status.idle":"2022-08-11T14:06:53.330556Z","shell.execute_reply.started":"2022-08-11T14:06:53.322986Z","shell.execute_reply":"2022-08-11T14:06:53.329601Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#draw a bar plot of survival by Pclass\nsns.barplot(x=\"Pclass\", y=\"Survived\", data=train)\n\n#print percentage of people by Pclass that survived\nprint(\"Percentage of Pclass(1) who survived:\", train[\"Survived\"][train[\"Pclass\"] == 1].value_counts(normalize = True)[1]*100)\n\nprint(\"Percentage of Pclass(2) who survived:\", train[\"Survived\"][train[\"Pclass\"] == 2].value_counts(normalize = True)[1]*100)\n\nprint(\"Percentage of Pclass(3) who survived:\", train[\"Survived\"][train[\"Pclass\"] == 3].value_counts(normalize = True)[1]*100)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T14:06:53.382324Z","iopub.execute_input":"2022-08-11T14:06:53.382746Z","iopub.status.idle":"2022-08-11T14:06:53.650796Z","shell.execute_reply.started":"2022-08-11T14:06:53.382712Z","shell.execute_reply":"2022-08-11T14:06:53.649808Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#draw a bar plot for SibSp vs. survival\nsns.barplot(x=\"SibSp\", y=\"Survived\", data=train)\n\n#I won't be printing individual percent values for all of these.\nprint(\"Percentage of SibSp(0) who survived:\", train[\"Survived\"][train[\"SibSp\"] == 0].value_counts(normalize = True)[1]*100)\n\nprint(\"Percentage of SibSp(1) who survived:\", train[\"Survived\"][train[\"SibSp\"] == 1].value_counts(normalize = True)[1]*100)\n\nprint(\"Percentage of SibSp(2) who survived:\", train[\"Survived\"][train[\"SibSp\"] == 2].value_counts(normalize = True)[1]*100)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T14:06:53.652722Z","iopub.execute_input":"2022-08-11T14:06:53.653149Z","iopub.status.idle":"2022-08-11T14:06:54.035094Z","shell.execute_reply.started":"2022-08-11T14:06:53.653115Z","shell.execute_reply":"2022-08-11T14:06:54.034189Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#draw a bar plot for Parch vs. survival\nsns.barplot(x=\"Parch\", y=\"Survived\", data=train)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-11T14:06:54.036590Z","iopub.execute_input":"2022-08-11T14:06:54.036894Z","iopub.status.idle":"2022-08-11T14:06:54.389824Z","shell.execute_reply.started":"2022-08-11T14:06:54.036864Z","shell.execute_reply":"2022-08-11T14:06:54.388688Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#sort the ages into logical categories\ntrain[\"Age\"] = train[\"Age\"].fillna(-0.5)\ntest[\"Age\"] = test[\"Age\"].fillna(-0.5)\nbins = [-1, 0, 5, 12, 18, 24, 35, 60, np.inf]\nlabels = ['Unknown', 'Baby', 'Child', 'Teenager', 'Student', 'Young Adult', 'Adult', 'Senior']\ntrain['AgeGroup'] = pd.cut(train[\"Age\"], bins, labels = labels)\ntest['AgeGroup'] = pd.cut(test[\"Age\"], bins, labels = labels)\n\n#draw a bar plot of Age vs. survival\nsns.barplot(x=\"AgeGroup\", y=\"Survived\", data=train)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-11T14:06:54.392360Z","iopub.execute_input":"2022-08-11T14:06:54.392746Z","iopub.status.idle":"2022-08-11T14:06:54.819079Z","shell.execute_reply.started":"2022-08-11T14:06:54.392710Z","shell.execute_reply":"2022-08-11T14:06:54.817924Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train[\"CabinBool\"] = (train[\"Cabin\"].notnull().astype('int'))\ntest[\"CabinBool\"] = (test[\"Cabin\"].notnull().astype('int'))\n\n#calculate percentages of CabinBool vs. survived\nprint(\"Percentage of CabinBool = 1 who survived:\", train[\"Survived\"][train[\"CabinBool\"] == 1].value_counts(normalize = True)[1]*100)\n\nprint(\"Percentage of CabinBool = 0 who survived:\", train[\"Survived\"][train[\"CabinBool\"] == 0].value_counts(normalize = True)[1]*100)\n#draw a bar plot of CabinBool vs. survival\nsns.barplot(x=\"CabinBool\", y=\"Survived\", data=train)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-11T14:06:54.820721Z","iopub.execute_input":"2022-08-11T14:06:54.821069Z","iopub.status.idle":"2022-08-11T14:06:55.069632Z","shell.execute_reply.started":"2022-08-11T14:06:54.821036Z","shell.execute_reply":"2022-08-11T14:06:55.068516Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.describe(include=\"all\")\n\n# We have a total of 418 passengers.\n# 1 value from the Fare feature is missing.\n# Around 20.5% of the Age feature is missing, we will need to fill that in","metadata":{"execution":{"iopub.status.busy":"2022-08-11T14:06:55.070920Z","iopub.execute_input":"2022-08-11T14:06:55.071852Z","iopub.status.idle":"2022-08-11T14:06:55.122741Z","shell.execute_reply.started":"2022-08-11T14:06:55.071816Z","shell.execute_reply":"2022-08-11T14:06:55.121481Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Drop Cabin as it won't be useful information\ntrain = train.drop(['Cabin'], axis = 1)\ntest = test.drop(['Cabin'], axis = 1)\n#Drop Ticket as it won't be useful information\ntrain = train.drop(['Ticket'], axis = 1)\ntest = test.drop(['Ticket'], axis = 1)\n\ntest.describe(include=\"all\")","metadata":{"execution":{"iopub.status.busy":"2022-08-11T14:06:55.124223Z","iopub.execute_input":"2022-08-11T14:06:55.124566Z","iopub.status.idle":"2022-08-11T14:06:55.174196Z","shell.execute_reply.started":"2022-08-11T14:06:55.124513Z","shell.execute_reply":"2022-08-11T14:06:55.172774Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#now we need to fill in the missing values in the Embarked feature\nprint(\"Number of people embarking in Southampton (S):\")\nsouthampton = train[train[\"Embarked\"] == \"S\"].shape[0]\nprint(southampton)\n\nprint(\"Number of people embarking in Cherbourg (C):\")\ncherbourg = train[train[\"Embarked\"] == \"C\"].shape[0]\nprint(cherbourg)\n\nprint(\"Number of people embarking in Queenstown (Q):\")\nqueenstown = train[train[\"Embarked\"] == \"Q\"].shape[0]\nprint(queenstown)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T14:06:55.175847Z","iopub.execute_input":"2022-08-11T14:06:55.176574Z","iopub.status.idle":"2022-08-11T14:06:55.189108Z","shell.execute_reply.started":"2022-08-11T14:06:55.176501Z","shell.execute_reply":"2022-08-11T14:06:55.187811Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#replacing the missing values in the Embarked feature with S\ntrain = train.fillna({\"Embarked\": \"S\"})","metadata":{"execution":{"iopub.status.busy":"2022-08-11T14:06:55.190727Z","iopub.execute_input":"2022-08-11T14:06:55.191239Z","iopub.status.idle":"2022-08-11T14:06:55.201697Z","shell.execute_reply.started":"2022-08-11T14:06:55.191202Z","shell.execute_reply":"2022-08-11T14:06:55.200444Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#create a combined group of both datasets\ncombine = [train, test]\n\n#extract a title for each Name in the train and test datasets\nfor dataset in combine:\n    dataset['Title'] = dataset.Name.str.extract(' ([A-Za-z]+)\\.', expand=False)\n\npd.crosstab(train['Title'], train['Sex'])","metadata":{"execution":{"iopub.status.busy":"2022-08-11T14:06:55.206821Z","iopub.execute_input":"2022-08-11T14:06:55.207204Z","iopub.status.idle":"2022-08-11T14:06:55.237881Z","shell.execute_reply.started":"2022-08-11T14:06:55.207169Z","shell.execute_reply":"2022-08-11T14:06:55.236650Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#replace various titles with more common names\nfor dataset in combine:\n    dataset['Title'] = dataset['Title'].replace(['Lady', 'Capt', 'Col',\n    'Don', 'Dr', 'Major', 'Rev', 'Jonkheer', 'Dona'], 'Rare')\n    \n    dataset['Title'] = dataset['Title'].replace(['Countess', 'Lady', 'Sir'], 'Royal')\n    dataset['Title'] = dataset['Title'].replace('Mlle', 'Miss')\n    dataset['Title'] = dataset['Title'].replace('Ms', 'Miss')\n    dataset['Title'] = dataset['Title'].replace('Mme', 'Mrs')\n\ntrain[['Title', 'Survived']].groupby(['Title'], as_index=False).mean()","metadata":{"execution":{"iopub.status.busy":"2022-08-11T14:06:55.239136Z","iopub.execute_input":"2022-08-11T14:06:55.239489Z","iopub.status.idle":"2022-08-11T14:06:55.265740Z","shell.execute_reply.started":"2022-08-11T14:06:55.239457Z","shell.execute_reply":"2022-08-11T14:06:55.264615Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#map each of the title groups to a numerical value\ntitle_mapping = {\"Mr\": 1, \"Miss\": 2, \"Mrs\": 3, \"Master\": 4, \"Royal\": 5, \"Rare\": 6}\nfor dataset in combine:\n    dataset['Title'] = dataset['Title'].map(title_mapping)\n    dataset['Title'] = dataset['Title'].fillna(0)\n\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-11T14:06:55.267286Z","iopub.execute_input":"2022-08-11T14:06:55.267675Z","iopub.status.idle":"2022-08-11T14:06:55.292298Z","shell.execute_reply.started":"2022-08-11T14:06:55.267629Z","shell.execute_reply":"2022-08-11T14:06:55.291478Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# fill missing age with mode age group for each title\nmr_age = train[train[\"Title\"] == 1][\"AgeGroup\"].mode() #Young Adult\nmiss_age = train[train[\"Title\"] == 2][\"AgeGroup\"].mode() #Student\nmrs_age = train[train[\"Title\"] == 3][\"AgeGroup\"].mode() #Adult\nmaster_age = train[train[\"Title\"] == 4][\"AgeGroup\"].mode() #Baby\nroyal_age = train[train[\"Title\"] == 5][\"AgeGroup\"].mode() #Adult\nrare_age = train[train[\"Title\"] == 6][\"AgeGroup\"].mode() #Adult\n\nage_title_mapping = {1: \"Young Adult\", 2: \"Student\", 3: \"Adult\", 4: \"Baby\", 5: \"Adult\", 6: \"Adult\"}\n\nfor x in range(len(train[\"AgeGroup\"])):\n    if train[\"AgeGroup\"][x] == \"Unknown\":\n        train[\"AgeGroup\"][x] = age_title_mapping[train[\"Title\"][x]]\n        \nfor x in range(len(test[\"AgeGroup\"])):\n    if test[\"AgeGroup\"][x] == \"Unknown\":\n        test[\"AgeGroup\"][x] = age_title_mapping[test[\"Title\"][x]]","metadata":{"execution":{"iopub.status.busy":"2022-08-11T14:06:55.293446Z","iopub.execute_input":"2022-08-11T14:06:55.294055Z","iopub.status.idle":"2022-08-11T14:06:55.440287Z","shell.execute_reply.started":"2022-08-11T14:06:55.294019Z","shell.execute_reply":"2022-08-11T14:06:55.439065Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#map each Age value to a numerical value\nage_mapping = {'Baby': 1, 'Child': 2, 'Teenager': 3, 'Student': 4, 'Young Adult': 5, 'Adult': 6, 'Senior': 7}\ntrain['AgeGroup'] = train['AgeGroup'].map(age_mapping)\ntest['AgeGroup'] = test['AgeGroup'].map(age_mapping)\n\ntrain.head()\n\n#dropping the Age feature for now, might change\ntrain = train.drop(['Age'], axis = 1)\ntest = test.drop(['Age'], axis = 1)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T14:06:55.441597Z","iopub.execute_input":"2022-08-11T14:06:55.441951Z","iopub.status.idle":"2022-08-11T14:06:55.457260Z","shell.execute_reply.started":"2022-08-11T14:06:55.441915Z","shell.execute_reply":"2022-08-11T14:06:55.455587Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#drop the name feature since it contains no more useful information.\ntrain = train.drop(['Name'], axis = 1)\ntest = test.drop(['Name'], axis = 1)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T14:06:55.458375Z","iopub.execute_input":"2022-08-11T14:06:55.458750Z","iopub.status.idle":"2022-08-11T14:06:55.470139Z","shell.execute_reply.started":"2022-08-11T14:06:55.458716Z","shell.execute_reply":"2022-08-11T14:06:55.468974Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#map each Sex value to a numerical value\nsex_mapping = {\"male\": 0, \"female\": 1}\ntrain['Sex'] = train['Sex'].map(sex_mapping)\ntest['Sex'] = test['Sex'].map(sex_mapping)\n\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-11T14:06:55.471665Z","iopub.execute_input":"2022-08-11T14:06:55.471994Z","iopub.status.idle":"2022-08-11T14:06:55.495201Z","shell.execute_reply.started":"2022-08-11T14:06:55.471963Z","shell.execute_reply":"2022-08-11T14:06:55.493963Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#map each Embarked value to a numerical value\nembarked_mapping = {\"S\": 1, \"C\": 2, \"Q\": 3}\ntrain['Embarked'] = train['Embarked'].map(embarked_mapping)\ntest['Embarked'] = test['Embarked'].map(embarked_mapping)\n\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-11T14:06:55.496747Z","iopub.execute_input":"2022-08-11T14:06:55.497113Z","iopub.status.idle":"2022-08-11T14:06:55.516815Z","shell.execute_reply.started":"2022-08-11T14:06:55.497079Z","shell.execute_reply":"2022-08-11T14:06:55.515993Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#fill in missing Fare value in test set based on mean fare for that Pclass \nfor x in range(len(test[\"Fare\"])):\n    if pd.isnull(test[\"Fare\"][x]):\n        pclass = test[\"Pclass\"][x] #Pclass = 3\n        test[\"Fare\"][x] = round(train[train[\"Pclass\"] == pclass][\"Fare\"].mean(), 4)\n        \n#map Fare values into groups of numerical values\ntrain['FareBand'] = pd.qcut(train['Fare'], 4, labels = [1, 2, 3, 4])\ntest['FareBand'] = pd.qcut(test['Fare'], 4, labels = [1, 2, 3, 4])\n\n#drop Fare values\ntrain = train.drop(['Fare'], axis = 1)\ntest = test.drop(['Fare'], axis = 1)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T14:06:55.517955Z","iopub.execute_input":"2022-08-11T14:06:55.518640Z","iopub.status.idle":"2022-08-11T14:06:55.543535Z","shell.execute_reply.started":"2022-08-11T14:06:55.518606Z","shell.execute_reply":"2022-08-11T14:06:55.542635Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#check train data\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-11T14:06:55.545283Z","iopub.execute_input":"2022-08-11T14:06:55.546425Z","iopub.status.idle":"2022-08-11T14:06:55.567528Z","shell.execute_reply.started":"2022-08-11T14:06:55.546377Z","shell.execute_reply":"2022-08-11T14:06:55.566354Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#check test data\ntest.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-11T14:06:55.569384Z","iopub.execute_input":"2022-08-11T14:06:55.569858Z","iopub.status.idle":"2022-08-11T14:06:55.598981Z","shell.execute_reply.started":"2022-08-11T14:06:55.569812Z","shell.execute_reply":"2022-08-11T14:06:55.597704Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\npredictors = train.drop(['Survived', 'PassengerId'], axis=1)\ntarget = train[\"Survived\"]\nx_train, x_val, y_train, y_val = train_test_split(predictors, target, test_size = 0.22, random_state = 0)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T14:06:55.600760Z","iopub.execute_input":"2022-08-11T14:06:55.601698Z","iopub.status.idle":"2022-08-11T14:06:55.612082Z","shell.execute_reply.started":"2022-08-11T14:06:55.601647Z","shell.execute_reply":"2022-08-11T14:06:55.611184Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Logistic Regression\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.metrics import accuracy_score\n\nlogreg = LogisticRegression()\nlogreg.fit(x_train, y_train)\ny_pred = logreg.predict(x_val)\nacc_logreg = round(accuracy_score(y_pred, y_val) * 100, 2)\nprint(acc_logreg)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T14:06:55.613938Z","iopub.execute_input":"2022-08-11T14:06:55.614722Z","iopub.status.idle":"2022-08-11T14:06:55.651170Z","shell.execute_reply.started":"2022-08-11T14:06:55.614672Z","shell.execute_reply":"2022-08-11T14:06:55.649953Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Support Vector Machines\nfrom sklearn.svm import SVC\n\nsvc = SVC()\nsvc.fit(x_train, y_train)\ny_pred = svc.predict(x_val)\nacc_svc = round(accuracy_score(y_pred, y_val) * 100, 2)\nprint(acc_svc)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T14:06:55.652904Z","iopub.execute_input":"2022-08-11T14:06:55.653279Z","iopub.status.idle":"2022-08-11T14:06:55.685129Z","shell.execute_reply.started":"2022-08-11T14:06:55.653244Z","shell.execute_reply":"2022-08-11T14:06:55.684120Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Decision Tree\nfrom sklearn.tree import DecisionTreeClassifier\n\ndecisiontree = DecisionTreeClassifier()\ndecisiontree.fit(x_train, y_train)\ny_pred = decisiontree.predict(x_val)\nacc_decisiontree = round(accuracy_score(y_pred, y_val) * 100, 2)\nprint(acc_decisiontree)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T14:06:55.686586Z","iopub.execute_input":"2022-08-11T14:06:55.686930Z","iopub.status.idle":"2022-08-11T14:06:55.700287Z","shell.execute_reply.started":"2022-08-11T14:06:55.686896Z","shell.execute_reply":"2022-08-11T14:06:55.699143Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Random Forest\nfrom sklearn.ensemble import RandomForestClassifier\n\nrandomforest = RandomForestClassifier()\nrandomforest.fit(x_train, y_train)\ny_pred = randomforest.predict(x_val)\nacc_randomforest = round(accuracy_score(y_pred, y_val) * 100, 2)\nprint(acc_randomforest)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T14:06:55.701974Z","iopub.execute_input":"2022-08-11T14:06:55.703181Z","iopub.status.idle":"2022-08-11T14:06:55.914409Z","shell.execute_reply.started":"2022-08-11T14:06:55.703129Z","shell.execute_reply":"2022-08-11T14:06:55.913324Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# KNN\nfrom sklearn.neighbors import KNeighborsClassifier\n\nknn = KNeighborsClassifier()\nknn.fit(x_train, y_train)\ny_pred = knn.predict(x_val)\nacc_knn = round(accuracy_score(y_pred, y_val) * 100, 2)\nprint(acc_knn)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T14:06:55.916262Z","iopub.execute_input":"2022-08-11T14:06:55.916658Z","iopub.status.idle":"2022-08-11T14:06:55.938623Z","shell.execute_reply.started":"2022-08-11T14:06:55.916623Z","shell.execute_reply":"2022-08-11T14:06:55.937303Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Stochastic Gradient Descent\nfrom sklearn.linear_model import SGDClassifier\n\nsgd = SGDClassifier()\nsgd.fit(x_train, y_train)\ny_pred = sgd.predict(x_val)\nacc_sgd = round(accuracy_score(y_pred, y_val) * 100, 2)\nprint(acc_sgd)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T14:06:55.940485Z","iopub.execute_input":"2022-08-11T14:06:55.941296Z","iopub.status.idle":"2022-08-11T14:06:55.960308Z","shell.execute_reply.started":"2022-08-11T14:06:55.941244Z","shell.execute_reply":"2022-08-11T14:06:55.959269Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"models = pd.DataFrame({\n    'Model': ['Support Vector Machines', 'KNN', 'Logistic Regression', \n              'Random Forest', 'Decision Tree', 'Stochastic Gradient Descent'],\n    'Score': [acc_svc, acc_knn, acc_logreg, \n              acc_randomforest, acc_decisiontree,\n              acc_sgd]})\nmodels.sort_values(by='Score', ascending=False)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T14:06:55.964496Z","iopub.execute_input":"2022-08-11T14:06:55.964898Z","iopub.status.idle":"2022-08-11T14:06:55.980602Z","shell.execute_reply.started":"2022-08-11T14:06:55.964860Z","shell.execute_reply":"2022-08-11T14:06:55.979749Z"},"trusted":true},"execution_count":null,"outputs":[]}]}