{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"In this capstone, I will be using the Titanic dataset to predit survived using the keyfeatures presented in the data set. \nI will go through data cleaning, feature selection/elimination/combination and then use some of the models learned in the class modules to see which model yields good results.\nAlso I will be tuning some of the hyperparamters for few models to see which one does the best job.\n","metadata":{}},{"cell_type":"code","source":"#Import basic libraries :\nimport pandas as pd\nimport numpy as np\nimport random as rnd\n\n# visualization\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n%matplotlib inline\n\n# machine learning\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.svm import SVC, LinearSVC\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.neighbors import KNeighborsClassifier\nfrom sklearn.tree import DecisionTreeClassifier","metadata":{"_uuid":"847a9b3972a6be2d2f3346ff01fea976d92ecdb6","_cell_guid":"5767a33c-8f18-4034-e52d-bf7a8f7d8ab8","execution":{"iopub.status.busy":"2022-08-11T20:04:00.756085Z","iopub.execute_input":"2022-08-11T20:04:00.756579Z","iopub.status.idle":"2022-08-11T20:04:00.764531Z","shell.execute_reply.started":"2022-08-11T20:04:00.756533Z","shell.execute_reply":"2022-08-11T20:04:00.763644Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Using Titanic dataset from Kaggle:\n#train_df = pd.read_csv('../input/train.csv')\n#test_df = pd.read_csv('../input/test.csv')\ntrain_df = pd.read_csv('https://raw.githubusercontent.com/pcsanwald/kaggle-titanic/master/train.csv')\ntest_df = pd.read_csv('https://raw.githubusercontent.com/pcsanwald/kaggle-titanic/master/test.csv')\ncombine = [train_df, test_df]\n\nprint(train_df.columns.values)\nprint(test_df.columns.values)\n","metadata":{"_uuid":"13f38775c12ad6f914254a08f0d1ef948a2bd453","_cell_guid":"e7319668-86fe-8adc-438d-0eef3fd0a982","execution":{"iopub.status.busy":"2022-08-11T20:04:00.765521Z","iopub.execute_input":"2022-08-11T20:04:00.765877Z","iopub.status.idle":"2022-08-11T20:04:01.995673Z","shell.execute_reply.started":"2022-08-11T20:04:00.765837Z","shell.execute_reply":"2022-08-11T20:04:01.994672Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# preview the data\ntrain_df.head()\ntrain_df.tail()\ntrain_df.info()\nprint('_'*40)\ntest_df.info()\ntrain_df.describe()\n","metadata":{"_uuid":"e068cd3a0465b65a0930a100cb348b9146d5fd2f","_cell_guid":"8d7ac195-ac1a-30a4-3f3f-80b8cf2c1c0f","execution":{"iopub.status.busy":"2022-08-11T20:04:01.997002Z","iopub.execute_input":"2022-08-11T20:04:01.997540Z","iopub.status.idle":"2022-08-11T20:04:02.048012Z","shell.execute_reply.started":"2022-08-11T20:04:01.997464Z","shell.execute_reply":"2022-08-11T20:04:02.047283Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Visualizing data and Correlating numerical features\n# grid = sns.FacetGrid(train_df, col='Pclass', hue='Survived')\ngrid = sns.FacetGrid(train_df, col='survived', row='pclass', height=2.2, aspect=1.6)\ngrid.map(plt.hist, 'age', alpha=.5, bins=20)\ngrid.add_legend();\nprint(train_df.columns.values)","metadata":{"_uuid":"d3a1fa63e9dd4f8a810086530a6363c94b36d030","_cell_guid":"50294eac-263a-af78-cb7e-3778eb9ad41f","execution":{"iopub.status.busy":"2022-08-11T20:04:02.049410Z","iopub.execute_input":"2022-08-11T20:04:02.049703Z","iopub.status.idle":"2022-08-11T20:04:03.184631Z","shell.execute_reply.started":"2022-08-11T20:04:02.049649Z","shell.execute_reply":"2022-08-11T20:04:03.183937Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Correlating categorical features, add Sex and Embarked features\n# grid = sns.FacetGrid(train_df, col='Embarked')\ngrid = sns.FacetGrid(train_df, row='embarked', height=2.2, aspect=1.6)\ngrid.map(sns.pointplot, 'pclass', 'survived', 'sex', palette='deep')\ngrid.add_legend()\nprint(train_df.columns.values)","metadata":{"_uuid":"c0e1f01b3f58e8f31b938b0e5eb1733132edc8ad","_cell_guid":"db57aabd-0e26-9ff9-9ebd-56d401cdf6e8","execution":{"iopub.status.busy":"2022-08-11T20:04:03.185749Z","iopub.execute_input":"2022-08-11T20:04:03.186175Z","iopub.status.idle":"2022-08-11T20:04:04.337580Z","shell.execute_reply.started":"2022-08-11T20:04:03.186132Z","shell.execute_reply":"2022-08-11T20:04:04.336645Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Correlating categorical and numerical features\n# grid = sns.FacetGrid(train_df, col='Embarked', hue='Survived', palette={0: 'k', 1: 'w'})\ngrid = sns.FacetGrid(train_df, row='embarked', col='survived', height=2.2, aspect=1.6)\ngrid.map(sns.barplot, 'sex', 'fare', alpha=.5, ci=None)\ngrid.add_legend()\nprint(train_df.columns.values)","metadata":{"_uuid":"c8fd535ac1bc90127369027c2101dbc939db118e","_cell_guid":"a21f66ac-c30d-f429-cc64-1da5460d16a9","execution":{"iopub.status.busy":"2022-08-11T20:04:04.339146Z","iopub.execute_input":"2022-08-11T20:04:04.339791Z","iopub.status.idle":"2022-08-11T20:04:05.426971Z","shell.execute_reply.started":"2022-08-11T20:04:04.339729Z","shell.execute_reply":"2022-08-11T20:04:05.426035Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(train_df.columns.values)\nprint(test_df.columns.values)\n#Dropping unsignificant features\nprint(\"Before\", train_df.shape, test_df.shape, combine[0].shape, combine[1].shape)\ntrain_df = train_df.drop(['ticket', 'cabin'], axis=1)\ntest_df = test_df.drop(['ticket', 'cabin'], axis=1)\ncombine = [train_df, test_df]\n\nprint(\"After\", train_df.shape, test_df.shape, combine[0].shape, combine[1].shape)\nprint(train_df.columns.values)","metadata":{"_uuid":"e328d9882affedcfc4c167aa5bb1ac132547558c","_cell_guid":"da057efe-88f0-bf49-917b-bb2fec418ed9","execution":{"iopub.status.busy":"2022-08-11T20:04:05.428569Z","iopub.execute_input":"2022-08-11T20:04:05.429168Z","iopub.status.idle":"2022-08-11T20:04:05.445903Z","shell.execute_reply.started":"2022-08-11T20:04:05.429106Z","shell.execute_reply":"2022-08-11T20:04:05.444809Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Creating new feature extracting from existing\nfor dataset in combine:\n    dataset['Title'] = dataset.name.str.extract(' ([A-Za-z]+)\\.', expand=False)\n\npd.crosstab(train_df['Title'], train_df['sex'])","metadata":{"_uuid":"c916644bd151f3dc8fca900f656d415b4c55e2bc","_cell_guid":"df7f0cd4-992c-4a79-fb19-bf6f0c024d4b","execution":{"iopub.status.busy":"2022-08-11T20:04:05.447776Z","iopub.execute_input":"2022-08-11T20:04:05.448369Z","iopub.status.idle":"2022-08-11T20:04:05.485583Z","shell.execute_reply.started":"2022-08-11T20:04:05.448061Z","shell.execute_reply":"2022-08-11T20:04:05.484634Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Replace many titles with a more common name or classify them as Rare\nfor dataset in combine:\n    dataset['Title'] = dataset['Title'].replace(['Lady', 'Countess','Capt', 'Col',\\\n \t'Don', 'Dr', 'Major', 'Rev', 'Sir', 'Jonkheer', 'Dona'], 'Rare')\n\n    dataset['Title'] = dataset['Title'].replace('Mlle', 'Miss')\n    dataset['Title'] = dataset['Title'].replace('Ms', 'Miss')\n    dataset['Title'] = dataset['Title'].replace('Mme', 'Mrs')\n    \ntrain_df[['Title', 'survived']].groupby(['Title'], as_index=False).mean()","metadata":{"_uuid":"b8cd938fba61fb4e226c77521b012f4bb8aa01d0","_cell_guid":"553f56d7-002a-ee63-21a4-c0efad10cfe9","execution":{"iopub.status.busy":"2022-08-11T20:04:05.487027Z","iopub.execute_input":"2022-08-11T20:04:05.487370Z","iopub.status.idle":"2022-08-11T20:04:05.514621Z","shell.execute_reply.started":"2022-08-11T20:04:05.487299Z","shell.execute_reply":"2022-08-11T20:04:05.513652Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Converting and fillin data:\ntitle_mapping = {\"Mr\": 1, \"Miss\": 2, \"Mrs\": 3, \"Master\": 4, \"Rare\": 5}\nfor dataset in combine:\n    dataset['Title'] = dataset['Title'].map(title_mapping)\n    dataset['Title'] = dataset['Title'].fillna(0)\n\ntrain_df.head()","metadata":{"_uuid":"e805ad52f0514497b67c3726104ba46d361eb92c","_cell_guid":"67444ebc-4d11-bac1-74a6-059133b6e2e8","execution":{"iopub.status.busy":"2022-08-11T20:04:05.515947Z","iopub.execute_input":"2022-08-11T20:04:05.516297Z","iopub.status.idle":"2022-08-11T20:04:05.549625Z","shell.execute_reply.started":"2022-08-11T20:04:05.516228Z","shell.execute_reply":"2022-08-11T20:04:05.548442Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Also drop Name and passengerid features from training and testing datasets. \ntrain_df = train_df.drop(['name'], axis=1)\ntest_df = test_df.drop(['name'], axis=1)\ncombine = [train_df, test_df]\ntrain_df.shape, test_df.shape","metadata":{"_uuid":"1da299cf2ffd399fd5b37d74fb40665d16ba5347","_cell_guid":"9d61dded-5ff0-5018-7580-aecb4ea17506","execution":{"iopub.status.busy":"2022-08-11T20:04:05.550967Z","iopub.execute_input":"2022-08-11T20:04:05.551239Z","iopub.status.idle":"2022-08-11T20:04:05.561462Z","shell.execute_reply.started":"2022-08-11T20:04:05.551183Z","shell.execute_reply":"2022-08-11T20:04:05.560587Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Dummy sex column:\nfor dataset in combine:\n   dataset['sex'] = dataset['sex'].map( {'female': 1, 'male': 0} ).astype(int)\n\ntrain_df.head()\n\n","metadata":{"_uuid":"840498eaee7baaca228499b0a5652da9d4edaf37","_cell_guid":"c20c1df2-157c-e5a0-3e24-15a828095c96","execution":{"iopub.status.busy":"2022-08-11T20:04:05.562535Z","iopub.execute_input":"2022-08-11T20:04:05.563084Z","iopub.status.idle":"2022-08-11T20:04:05.591332Z","shell.execute_reply.started":"2022-08-11T20:04:05.563038Z","shell.execute_reply":"2022-08-11T20:04:05.590395Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Guess missing age data based on Pclass x Gender combinations.\nguess_ages = np.zeros((2,3))\nguess_ages\nfor dataset in combine:\n    for i in range(0, 2):\n        for j in range(0, 3):\n            guess_df = dataset[(dataset['sex'] == i) & \\\n                                  (dataset['pclass'] == j+1)]['age'].dropna()\n\n            # age_mean = guess_df.mean()\n            # age_std = guess_df.std()\n            # age_guess = rnd.uniform(age_mean - age_std, age_mean + age_std)\n\n            age_guess = guess_df.median()\n\n            # Convert random age float to nearest .5 age\n            guess_ages[i,j] = int( age_guess/0.5 + 0.5 ) * 0.5\n            \n    for i in range(0, 2):\n        for j in range(0, 3):\n            dataset.loc[ (dataset.age.isnull()) & (dataset.sex == i) & (dataset.pclass == j+1),\\\n                    'age'] = guess_ages[i,j]\n\n    dataset['age'] = dataset['age'].astype(int)\n\ntrain_df.head()","metadata":{"_uuid":"31198f0ad0dbbb74290ebe135abffa994b8f58f3","_cell_guid":"a4015dfa-a0ab-65bc-0cbe-efecf1eb2569","execution":{"iopub.status.busy":"2022-08-11T20:04:05.592736Z","iopub.execute_input":"2022-08-11T20:04:05.593018Z","iopub.status.idle":"2022-08-11T20:04:05.689684Z","shell.execute_reply.started":"2022-08-11T20:04:05.592959Z","shell.execute_reply":"2022-08-11T20:04:05.688906Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#create Age bands and determine correlations with Survived\ntrain_df['AgeBand'] = pd.cut(train_df['age'], 5)\ntrain_df[['AgeBand', 'survived']].groupby(['AgeBand'], as_index=False).mean().sort_values(by='AgeBand', ascending=True)","metadata":{"_uuid":"5c8b4cbb302f439ef0d6278dcfbdafd952675353","_cell_guid":"725d1c84-6323-9d70-5812-baf9994d3aa1","execution":{"iopub.status.busy":"2022-08-11T20:04:05.690943Z","iopub.execute_input":"2022-08-11T20:04:05.691415Z","iopub.status.idle":"2022-08-11T20:04:05.715903Z","shell.execute_reply.started":"2022-08-11T20:04:05.691370Z","shell.execute_reply":"2022-08-11T20:04:05.714998Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#replace Age based on AgeBand.\nfor dataset in combine:    \n    dataset.loc[ dataset['age'] <= 16, 'age'] = 0\n    dataset.loc[(dataset['age'] > 16) & (dataset['age'] <= 32), 'age'] = 1\n    dataset.loc[(dataset['age'] > 32) & (dataset['age'] <= 48), 'age'] = 2\n    dataset.loc[(dataset['age'] > 48) & (dataset['age'] <= 64), 'age'] = 3\n    dataset.loc[ dataset['age'] > 64, 'age']\ntrain_df.head()","metadata":{"_uuid":"ee13831345f389db407c178f66c19cc8331445b0","_cell_guid":"797b986d-2c45-a9ee-e5b5-088de817c8b2","execution":{"iopub.status.busy":"2022-08-11T20:04:05.717099Z","iopub.execute_input":"2022-08-11T20:04:05.717361Z","iopub.status.idle":"2022-08-11T20:04:05.773405Z","shell.execute_reply.started":"2022-08-11T20:04:05.717309Z","shell.execute_reply":"2022-08-11T20:04:05.772633Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#We can now remove the AgeBand feature.\ntrain_df = train_df.drop(['AgeBand'], axis=1)\ncombine = [train_df, test_df]\ntrain_df.head()","metadata":{"_uuid":"1ea01ccc4a24e8951556d97c990aa0136da19721","_cell_guid":"875e55d4-51b0-5061-b72c-8a23946133a3","execution":{"iopub.status.busy":"2022-08-11T20:04:05.775033Z","iopub.execute_input":"2022-08-11T20:04:05.775289Z","iopub.status.idle":"2022-08-11T20:04:05.798400Z","shell.execute_reply.started":"2022-08-11T20:04:05.775240Z","shell.execute_reply":"2022-08-11T20:04:05.797532Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Create a new feature for FamilySize which combines Parch and SibSp. This will enable us to drop Parch and SibSp from our datasets.\nfor dataset in combine:\n    dataset['FamilySize'] = dataset['sibsp'] + dataset['parch'] + 1\n\ntrain_df[['FamilySize', 'survived']].groupby(['FamilySize'], as_index=False).mean().sort_values(by='survived', ascending=False)","metadata":{"_uuid":"33d1236ce4a8ab888b9fac2d5af1c78d174b32c7","_cell_guid":"7e6c04ed-cfaa-3139-4378-574fd095d6ba","execution":{"iopub.status.busy":"2022-08-11T20:04:05.799624Z","iopub.execute_input":"2022-08-11T20:04:05.799898Z","iopub.status.idle":"2022-08-11T20:04:05.823369Z","shell.execute_reply.started":"2022-08-11T20:04:05.799845Z","shell.execute_reply":"2022-08-11T20:04:05.822630Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#create another feature called IsAlone\nfor dataset in combine:\n    dataset['IsAlone'] = 0\n    dataset.loc[dataset['FamilySize'] == 1, 'IsAlone'] = 1\n\ntrain_df[['IsAlone', 'survived']].groupby(['IsAlone'], as_index=False).mean()","metadata":{"_uuid":"3b8db81cc3513b088c6bcd9cd1938156fe77992f","_cell_guid":"5c778c69-a9ae-1b6b-44fe-a0898d07be7a","execution":{"iopub.status.busy":"2022-08-11T20:04:05.824587Z","iopub.execute_input":"2022-08-11T20:04:05.824853Z","iopub.status.idle":"2022-08-11T20:04:05.854381Z","shell.execute_reply.started":"2022-08-11T20:04:05.824803Z","shell.execute_reply":"2022-08-11T20:04:05.853693Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#drop Parch, SibSp, and FamilySize features in favor of IsAlone.\ntrain_df = train_df.drop(['parch', 'sibsp', 'FamilySize'], axis=1)\ntest_df = test_df.drop(['parch', 'sibsp', 'FamilySize'], axis=1)\ncombine = [train_df, test_df]\n\ntrain_df.head()","metadata":{"_uuid":"1e3479690ef7cd8ee10538d4f39d7117246887f0","_cell_guid":"74ee56a6-7357-f3bc-b605-6c41f8aa6566","execution":{"iopub.status.busy":"2022-08-11T20:04:05.855572Z","iopub.execute_input":"2022-08-11T20:04:05.855833Z","iopub.status.idle":"2022-08-11T20:04:05.879453Z","shell.execute_reply.started":"2022-08-11T20:04:05.855782Z","shell.execute_reply":"2022-08-11T20:04:05.878751Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#create an artificial feature combining Pclass and Age\nfor dataset in combine:\n    dataset['age*class'] = dataset.age * dataset.pclass\n\ntrain_df.loc[:, ['age*class', 'age', 'pclass']].head(10)","metadata":{"_uuid":"aac2c5340c06210a8b0199e15461e9049fbf2cff","_cell_guid":"305402aa-1ea1-c245-c367-056eef8fe453","execution":{"iopub.status.busy":"2022-08-11T20:04:05.880584Z","iopub.execute_input":"2022-08-11T20:04:05.880798Z","iopub.status.idle":"2022-08-11T20:04:05.897630Z","shell.execute_reply.started":"2022-08-11T20:04:05.880763Z","shell.execute_reply":"2022-08-11T20:04:05.896930Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Embarked feature takes S, Q, C values based on port of embarkation. \nfreq_port = train_df.embarked.dropna().mode()[0]\nfreq_port","metadata":{"_uuid":"1e3f8af166f60a1b3125a6b046eff5fff02d63cf","_cell_guid":"bf351113-9b7f-ef56-7211-e8dd00665b18","execution":{"iopub.status.busy":"2022-08-11T20:04:05.898481Z","iopub.execute_input":"2022-08-11T20:04:05.898897Z","iopub.status.idle":"2022-08-11T20:04:05.912052Z","shell.execute_reply.started":"2022-08-11T20:04:05.898708Z","shell.execute_reply":"2022-08-11T20:04:05.911180Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for dataset in combine:\n    dataset['embarked'] = dataset['embarked'].fillna(freq_port)\n    \ntrain_df[['embarked', 'survived']].groupby(['embarked'], as_index=False).mean().sort_values(by='survived', ascending=False)","metadata":{"_uuid":"d85b5575fb45f25749298641f6a0a38803e1ff22","_cell_guid":"51c21fcc-f066-cd80-18c8-3d140be6cbae","execution":{"iopub.status.busy":"2022-08-11T20:04:05.912914Z","iopub.execute_input":"2022-08-11T20:04:05.913168Z","iopub.status.idle":"2022-08-11T20:04:05.937366Z","shell.execute_reply.started":"2022-08-11T20:04:05.913118Z","shell.execute_reply":"2022-08-11T20:04:05.936681Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Convert the EmbarkedFill feature by creating a new numeric Port feature\nfor dataset in combine:\n    dataset['embarked'] = dataset['embarked'].map( {'S': 0, 'C': 1, 'Q': 2} ).astype(int)\n\ntrain_df.head()","metadata":{"_uuid":"e480a1ef145de0b023821134896391d568a6f4f9","_cell_guid":"89a91d76-2cc0-9bbb-c5c5-3c9ecae33c66","execution":{"iopub.status.busy":"2022-08-11T20:04:05.938589Z","iopub.execute_input":"2022-08-11T20:04:05.938848Z","iopub.status.idle":"2022-08-11T20:04:05.964672Z","shell.execute_reply.started":"2022-08-11T20:04:05.938797Z","shell.execute_reply":"2022-08-11T20:04:05.964001Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Round off the fare to two decimals as it represents currency\ntest_df['fare'].fillna(test_df['fare'].dropna().median(), inplace=True)\ntest_df.head()","metadata":{"_uuid":"aacb62f3526072a84795a178bd59222378bab180","_cell_guid":"3600cb86-cf5f-d87b-1b33-638dc8db1564","execution":{"iopub.status.busy":"2022-08-11T20:04:05.965593Z","iopub.execute_input":"2022-08-11T20:04:05.965962Z","iopub.status.idle":"2022-08-11T20:04:05.988574Z","shell.execute_reply.started":"2022-08-11T20:04:05.965922Z","shell.execute_reply":"2022-08-11T20:04:05.987523Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#We can now create FareBand.\ntrain_df['FareBand'] = pd.qcut(train_df['fare'], 4)\ntrain_df[['FareBand', 'survived']].groupby(['FareBand'], as_index=False).mean().sort_values(by='FareBand', ascending=True)","metadata":{"_uuid":"b9a78f6b4c72520d4ad99d2c89c84c591216098d","_cell_guid":"0e9018b1-ced5-9999-8ce1-258a0952cbf2","execution":{"iopub.status.busy":"2022-08-11T20:04:05.989650Z","iopub.execute_input":"2022-08-11T20:04:05.990059Z","iopub.status.idle":"2022-08-11T20:04:06.016098Z","shell.execute_reply.started":"2022-08-11T20:04:05.989841Z","shell.execute_reply":"2022-08-11T20:04:06.015309Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Convert the Fare feature to ordinal values based on the FareBand\nfor dataset in combine:\n    dataset.loc[ dataset['fare'] <= 7.91, 'fare'] = 0\n    dataset.loc[(dataset['fare'] > 7.91) & (dataset['fare'] <= 14.454), 'fare'] = 1\n    dataset.loc[(dataset['fare'] > 14.454) & (dataset['fare'] <= 31), 'fare']   = 2\n    dataset.loc[ dataset['fare'] > 31, 'fare'] = 3\n    dataset['fare'] = dataset['fare'].astype(int)\n\ntrain_df = train_df.drop(['FareBand'], axis=1)\ncombine = [train_df, test_df]\n    \ntrain_df.head(10)","metadata":{"_uuid":"640f305061ec4221a45ba250f8d54bb391035a57","_cell_guid":"385f217a-4e00-76dc-1570-1de4eec0c29c","execution":{"iopub.status.busy":"2022-08-11T20:04:06.017420Z","iopub.execute_input":"2022-08-11T20:04:06.017773Z","iopub.status.idle":"2022-08-11T20:04:06.071987Z","shell.execute_reply.started":"2022-08-11T20:04:06.017711Z","shell.execute_reply":"2022-08-11T20:04:06.071004Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Exam the test dataset.\ntest_df.head(10)","metadata":{"_uuid":"8453cecad81fcc44de3f4e4e4c3ce6afa977740d","_cell_guid":"d2334d33-4fe5-964d-beac-6aa620066e15","execution":{"iopub.status.busy":"2022-08-11T20:04:06.073629Z","iopub.execute_input":"2022-08-11T20:04:06.074135Z","iopub.status.idle":"2022-08-11T20:04:06.094769Z","shell.execute_reply.started":"2022-08-11T20:04:06.073889Z","shell.execute_reply":"2022-08-11T20:04:06.093880Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#To identify relationship between output (Survived or not) with other variables or features (Gender, Age, Port...) and train our model with a given dataset:\n#Logistic Regression\n#KNN\n#Support Vector Machines\n#Decision Tree\n#Random Forrest\n\n\nX_train = train_df.drop(\"survived\", axis=1)\nY_train = train_df[\"survived\"]\nX_test  = test_df.copy()\nX_train.shape, Y_train.shape, X_test.shape","metadata":{"_uuid":"04d2235855f40cffd81f76b977a500fceaae87ad","_cell_guid":"0acf54f9-6cf5-24b5-72d9-29b30052823a","execution":{"iopub.status.busy":"2022-08-11T20:04:06.095974Z","iopub.execute_input":"2022-08-11T20:04:06.096234Z","iopub.status.idle":"2022-08-11T20:04:06.107800Z","shell.execute_reply.started":"2022-08-11T20:04:06.096193Z","shell.execute_reply":"2022-08-11T20:04:06.106854Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Logistic Regression\n\nlogreg = LogisticRegression()\nlogreg.fit(X_train, Y_train)\nY_pred = logreg.predict(X_test)\nacc_log = round(logreg.score(X_train, Y_train) * 100, 2)\nacc_log\n\ncoeff_df = pd.DataFrame(train_df.columns.delete(0))\ncoeff_df.columns = ['Feature']\ncoeff_df[\"Correlation\"] = pd.Series(logreg.coef_[0])\n\ncoeff_df.sort_values(by='Correlation', ascending=False)","metadata":{"_uuid":"a649b9c53f4c7b40694f60f5c8dc14ec5ef519ec","_cell_guid":"0edd9322-db0b-9c37-172d-a3a4f8dec229","execution":{"iopub.status.busy":"2022-08-11T20:04:06.109127Z","iopub.execute_input":"2022-08-11T20:04:06.109555Z","iopub.status.idle":"2022-08-11T20:04:06.137862Z","shell.execute_reply.started":"2022-08-11T20:04:06.109342Z","shell.execute_reply":"2022-08-11T20:04:06.136944Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"_uuid":"60039d5377da49f1aa9ac4a924331328bd69add1","_cell_guid":"7a63bf04-a410-9c81-5310-bdef7963298f","execution":{"iopub.status.busy":"2022-08-11T20:04:06.138973Z","iopub.execute_input":"2022-08-11T20:04:06.139331Z","iopub.status.idle":"2022-08-11T20:04:06.195452Z","shell.execute_reply.started":"2022-08-11T20:04:06.139165Z","shell.execute_reply":"2022-08-11T20:04:06.194628Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#KNN confidence score is better than Logistics Regression but worse than SVM\nknn = KNeighborsClassifier(n_neighbors = 3)\nknn.fit(X_train, Y_train)\nY_pred = knn.predict(X_test)\nacc_knn = round(knn.score(X_train, Y_train) * 100, 2)\nprint(acc_knn)\n\n#Turn KNN model by changing the n_neighbors value:\nknn5 = KNeighborsClassifier(n_neighbors = 5)\nknn5.fit(X_train, Y_train)\nY_pred = knn5.predict(X_test)\nacc_knn5 = round(knn5.score(X_train, Y_train) * 100, 2)\nprint(acc_knn5)","metadata":{"_uuid":"54d86cd45703d459d452f89572771deaa8877999","_cell_guid":"ca14ae53-f05e-eb73-201c-064d7c3ed610","execution":{"iopub.status.busy":"2022-08-11T20:06:40.789444Z","iopub.execute_input":"2022-08-11T20:06:40.789814Z","iopub.status.idle":"2022-08-11T20:06:40.825895Z","shell.execute_reply.started":"2022-08-11T20:06:40.789760Z","shell.execute_reply":"2022-08-11T20:06:40.825039Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Turns out in this case, n_neighbor value did not optimized the model.","metadata":{}},{"cell_type":"code","source":"#What about decrease the value of n_neighbor?\nknn2 = KNeighborsClassifier(n_neighbors = 2)\nknn2.fit(X_train, Y_train)\nY_pred = knn2.predict(X_test)\nacc_knn2 = round(knn2.score(X_train, Y_train) * 100, 2)\nprint(acc_knn2)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T20:08:48.679031Z","iopub.execute_input":"2022-08-11T20:08:48.679774Z","iopub.status.idle":"2022-08-11T20:08:48.700747Z","shell.execute_reply.started":"2022-08-11T20:08:48.679701Z","shell.execute_reply":"2022-08-11T20:08:48.699786Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"So seems the first choice of n_neighbors = 3 is the best value for this model","metadata":{}},{"cell_type":"code","source":"#Compare SVC linear kernel vs LinearSVC functions:\n\n# Linear SVC\nlinear_svc = LinearSVC()\nlinear_svc.fit(X_train, Y_train)\nY_pred = linear_svc.predict(X_test)\nacc_linear_svc = round(linear_svc.score(X_train, Y_train) * 100, 2)\nprint(acc_linear_svc)\n\n# Support Vector Machines\nsvclinear = SVC(kernel='linear')\nsvclinear.fit(X_train, Y_train)\nY_pred = svclinear.predict(X_test)\nacc_svclinear = round(svclinear.score(X_train, Y_train) * 100, 2)\nprint(acc_svclinear)\n\n#Tuning the SVC model using RBF kernel:\nsvcrbf = SVC(kernel='rbf', random_state=0, gamma=1, C=1)\nsvcrbf.fit(X_train, Y_train)\nY_pred = svcrbf.predict(X_test)\nacc_svcrbf = round(svcrbf.score(X_train, Y_train) * 100, 2)\nprint(acc_svcrbf)","metadata":{"_uuid":"52ea4f44dd626448dd2199cb284b592670b1394b","_cell_guid":"a4d56857-9432-55bb-14c0-52ebeb64d198","execution":{"iopub.status.busy":"2022-08-11T20:33:33.434935Z","iopub.execute_input":"2022-08-11T20:33:33.435627Z","iopub.status.idle":"2022-08-11T20:33:33.575746Z","shell.execute_reply.started":"2022-08-11T20:33:33.435565Z","shell.execute_reply":"2022-08-11T20:33:33.575002Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Seems Using the RBF kernel yield a better model","metadata":{}},{"cell_type":"code","source":"# Decision Tree\nimport time\nstart = time.time()\n\ndecision_tree = DecisionTreeClassifier()\ndecision_tree.fit(X_train, Y_train)\nY_pred = decision_tree.predict(X_test)\nacc_decision_tree = round(decision_tree.score(X_train, Y_train) * 100, 2)\nprint(acc_decision_tree)\n\n#print(f'Decision Tree runtime elapsed: {round(time.time() - start, 2)} seconds.')\n\n#Turn the model by set the max_depth to 3:\ndecision_tree3 = DecisionTreeClassifier(max_depth=3)\ndecision_tree3.fit(X_train, Y_train)\nY_pred = decision_tree3.predict(X_test)\nacc_decision_tree3 = round(decision_tree3.score(X_train, Y_train) * 100, 2)\nprint(acc_decision_tree3)","metadata":{"_uuid":"1f94308b23b934123c03067e84027b507b989e52","_cell_guid":"dd85f2b7-ace2-0306-b4ec-79c68cd3fea0","execution":{"iopub.status.busy":"2022-08-11T20:46:12.640216Z","iopub.execute_input":"2022-08-11T20:46:12.640818Z","iopub.status.idle":"2022-08-11T20:46:12.656842Z","shell.execute_reply.started":"2022-08-11T20:46:12.640664Z","shell.execute_reply":"2022-08-11T20:46:12.655692Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#Seems the default one yield a better result.","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Random Forest\n\nstart = time.time()\n\nrandom_forest = RandomForestClassifier(n_estimators=100)\nrandom_forest.fit(X_train, Y_train)\nY_pred = random_forest.predict(X_test)\nrandom_forest.score(X_train, Y_train)\nacc_random_forest = round(random_forest.score(X_train, Y_train) * 100, 2)\nacc_random_forest\n\nprint(f'Decision Tree runtime elapsed: {round(time.time() - start, 2)} seconds.')","metadata":{"_uuid":"483c647d2759a2703d20785a44f51b6dee47d0db","_cell_guid":"f0694a8e-b618-8ed9-6f0d-8c6fba2c4567","execution":{"iopub.status.busy":"2022-08-11T20:04:06.304167Z","iopub.execute_input":"2022-08-11T20:04:06.304626Z","iopub.status.idle":"2022-08-11T20:04:06.481533Z","shell.execute_reply.started":"2022-08-11T20:04:06.304582Z","shell.execute_reply":"2022-08-11T20:04:06.480659Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Model Evaluation:\nmodels = pd.DataFrame({\n    'Model': ['Support Vector Machines', 'KNN', 'Logistic Regression', \n              'Random Forest', 'Decision Tree'],\n    'Score': [acc_svc, acc_knn, acc_log, \n              acc_random_forest, acc_decision_tree]})\nmodels.sort_values(by='Score', ascending=False)","metadata":{"_uuid":"06a52babe50e0dd837b553c78fc73872168e1c7d","_cell_guid":"1f3cebe0-31af-70b2-1ce4-0fd406bcdfc6","execution":{"iopub.status.busy":"2022-08-11T20:04:06.482736Z","iopub.execute_input":"2022-08-11T20:04:06.482977Z","iopub.status.idle":"2022-08-11T20:04:06.497854Z","shell.execute_reply.started":"2022-08-11T20:04:06.482932Z","shell.execute_reply":"2022-08-11T20:04:06.497095Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Looks line Random Forest and Decision Tree models performed equally well.","metadata":{}}]}