{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# data analysis and wrangling\nimport pandas as pd\nimport numpy as np\nimport random as rnd\n\n# visualization\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n%matplotlib inline\n\n# machine learning\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.svm import SVC, LinearSVC\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.neighbors import KNeighborsClassifier\nfrom sklearn.naive_bayes import GaussianNB\nfrom sklearn.linear_model import Perceptron\nfrom sklearn.linear_model import SGDClassifier\nfrom sklearn.tree import DecisionTreeClassifier","metadata":{"_uuid":"847a9b3972a6be2d2f3346ff01fea976d92ecdb6","_cell_guid":"5767a33c-8f18-4034-e52d-bf7a8f7d8ab8","execution":{"iopub.status.busy":"2022-07-28T16:40:49.192899Z","iopub.execute_input":"2022-07-28T16:40:49.193589Z","iopub.status.idle":"2022-07-28T16:40:49.203518Z","shell.execute_reply.started":"2022-07-28T16:40:49.193518Z","shell.execute_reply":"2022-07-28T16:40:49.202785Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.read_csv('../input/train.csv')\ntest_df = pd.read_csv('../input/test.csv')\ncombine = [train_df, test_df]\n\nprint(train_df.columns.values)","metadata":{"_uuid":"13f38775c12ad6f914254a08f0d1ef948a2bd453","_cell_guid":"e7319668-86fe-8adc-438d-0eef3fd0a982","execution":{"iopub.status.busy":"2022-07-28T16:40:49.354192Z","iopub.execute_input":"2022-07-28T16:40:49.354661Z","iopub.status.idle":"2022-07-28T16:40:49.376473Z","shell.execute_reply.started":"2022-07-28T16:40:49.354617Z","shell.execute_reply":"2022-07-28T16:40:49.375590Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# preview the data\ntrain_df.head()\ntrain_df.tail()\n\ntrain_df.info()\nprint('_'*40)\ntest_df.info()\n\ntrain_df.describe()\n# Review survived rate using `percentiles=[.61, .62]` knowing our problem description mentions 38% survival rate.\n# Review Parch distribution using `percentiles=[.75, .8]`\n# SibSp distribution `[.68, .69]`\n# Age and Fare `[.1, .2, .3, .4, .5, .6, .7, .8, .9, .99]`\n\n","metadata":{"_uuid":"e068cd3a0465b65a0930a100cb348b9146d5fd2f","_cell_guid":"8d7ac195-ac1a-30a4-3f3f-80b8cf2c1c0f","execution":{"iopub.status.busy":"2022-07-28T16:40:49.500928Z","iopub.execute_input":"2022-07-28T16:40:49.501273Z","iopub.status.idle":"2022-07-28T16:40:49.563516Z","shell.execute_reply.started":"2022-07-28T16:40:49.501218Z","shell.execute_reply":"2022-07-28T16:40:49.562587Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Analyze by pivoting features\n#train_df[['Pclass', 'Survived']].groupby(['Pclass'], as_index=False).mean().sort_values(by='Survived', ascending=False)\n#train_df[[\"Sex\", \"Survived\"]].groupby(['Sex'], as_index=False).mean().sort_values(by='Survived', ascending=False)\n#train_df[[\"SibSp\", \"Survived\"]].groupby(['SibSp'], as_index=False).mean().sort_values(by='Survived', ascending=False)\n#train_df[[\"Parch\", \"Survived\"]].groupby(['Parch'], as_index=False).mean().sort_values(by='Survived', ascending=False)","metadata":{"_uuid":"97a845528ce9f76e85055a4bb9e97c27091f6aa1","_cell_guid":"0964832a-a4be-2d6f-a89e-63526389cee9","execution":{"iopub.status.busy":"2022-07-28T16:40:49.640580Z","iopub.execute_input":"2022-07-28T16:40:49.640905Z","iopub.status.idle":"2022-07-28T16:40:49.645751Z","shell.execute_reply.started":"2022-07-28T16:40:49.640853Z","shell.execute_reply":"2022-07-28T16:40:49.644298Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Visualizing data and Correlating numerical features\ng = sns.FacetGrid(train_df, col='Survived')\ng.map(plt.hist, 'Age', bins=20)\n\n# grid = sns.FacetGrid(train_df, col='Pclass', hue='Survived')\ngrid = sns.FacetGrid(train_df, col='Survived', row='Pclass', height=2.2, aspect=1.6)\ngrid.map(plt.hist, 'Age', alpha=.5, bins=20)\ngrid.add_legend();","metadata":{"_uuid":"d3a1fa63e9dd4f8a810086530a6363c94b36d030","_cell_guid":"50294eac-263a-af78-cb7e-3778eb9ad41f","execution":{"iopub.status.busy":"2022-07-28T16:40:49.864741Z","iopub.execute_input":"2022-07-28T16:40:49.865035Z","iopub.status.idle":"2022-07-28T16:40:52.266709Z","shell.execute_reply.started":"2022-07-28T16:40:49.864988Z","shell.execute_reply":"2022-07-28T16:40:52.265158Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Correlating categorical features\n\nNow we can correlate categorical features with our solution goal.\n\n**Observations.**\n\n- Female passengers had much better survival rate than males. Confirms classifying (#1).\n- Exception in Embarked=C where males had higher survival rate. This could be a correlation between Pclass and Embarked and in turn Pclass and Survived, not necessarily direct correlation between Embarked and Survived.\n- Males had better survival rate in Pclass=3 when compared with Pclass=2 for C and Q ports. Completing (#2).\n- Ports of embarkation have varying survival rates for Pclass=3 and among male passengers. Correlating (#1).\n\n**Decisions.**\n\n- Add Sex feature to model training.\n- Complete and add Embarked feature to model training.","metadata":{"_uuid":"892ab7ee88b1b1c5f1ac987884fa31e111bb0507","_cell_guid":"36f5a7c0-c55c-f76f-fdf8-945a32a68cb0"}},{"cell_type":"code","source":"#Correlating categorical features\n# grid = sns.FacetGrid(train_df, col='Embarked')\ngrid = sns.FacetGrid(train_df, row='Embarked', size=2.2, aspect=1.6)\ngrid.map(sns.pointplot, 'Pclass', 'Survived', 'Sex', palette='deep')\ngrid.add_legend()","metadata":{"_uuid":"c0e1f01b3f58e8f31b938b0e5eb1733132edc8ad","_cell_guid":"db57aabd-0e26-9ff9-9ebd-56d401cdf6e8","execution":{"iopub.status.busy":"2022-07-28T16:40:52.268844Z","iopub.execute_input":"2022-07-28T16:40:52.269515Z","iopub.status.idle":"2022-07-28T16:40:53.630674Z","shell.execute_reply.started":"2022-07-28T16:40:52.269239Z","shell.execute_reply":"2022-07-28T16:40:53.629329Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Correlating categorical and numerical features\n# grid = sns.FacetGrid(train_df, col='Embarked', hue='Survived', palette={0: 'k', 1: 'w'})\ngrid = sns.FacetGrid(train_df, row='Embarked', col='Survived', height=2.2, aspect=1.6)\ngrid.map(sns.barplot, 'Sex', 'Fare', alpha=.5, ci=None)\ngrid.add_legend()","metadata":{"_uuid":"c8fd535ac1bc90127369027c2101dbc939db118e","_cell_guid":"a21f66ac-c30d-f429-cc64-1da5460d16a9","execution":{"iopub.status.busy":"2022-07-28T16:40:53.632911Z","iopub.execute_input":"2022-07-28T16:40:53.633442Z","iopub.status.idle":"2022-07-28T16:40:55.136915Z","shell.execute_reply.started":"2022-07-28T16:40:53.633355Z","shell.execute_reply":"2022-07-28T16:40:55.135243Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Wrangle data\n\nWe have collected several assumptions and decisions regarding our datasets and solution requirements. So far we did not have to change a single feature or value to arrive at these. Let us now execute our decisions and assumptions for correcting, creating, and completing goals.\n\n### Correcting by dropping features\n\nThis is a good starting goal to execute. By dropping features we are dealing with fewer data points. Speeds up our notebook and eases the analysis.\n\nBased on our assumptions and decisions we want to drop the Cabin (correcting #2) and Ticket (correcting #1) features.\n\nNote that where applicable we perform operations on both training and testing datasets together to stay consistent.","metadata":{"_uuid":"73a9111a8dc2a6b8b6c78ef628b6cae2a63fc33f","_cell_guid":"cfac6291-33cc-506e-e548-6cad9408623d"}},{"cell_type":"code","source":"#Correcting by dropping features\nprint(\"Before\", train_df.shape, test_df.shape, combine[0].shape, combine[1].shape)\n\ntrain_df = train_df.drop(['Ticket', 'Cabin'], axis=1)\ntest_df = test_df.drop(['Ticket', 'Cabin'], axis=1)\ncombine = [train_df, test_df]\n\nprint(\"After\", train_df.shape, test_df.shape, combine[0].shape, combine[1].shape)","metadata":{"_uuid":"e328d9882affedcfc4c167aa5bb1ac132547558c","_cell_guid":"da057efe-88f0-bf49-917b-bb2fec418ed9","execution":{"iopub.status.busy":"2022-07-28T16:40:55.139228Z","iopub.execute_input":"2022-07-28T16:40:55.139882Z","iopub.status.idle":"2022-07-28T16:40:55.156856Z","shell.execute_reply.started":"2022-07-28T16:40:55.139603Z","shell.execute_reply":"2022-07-28T16:40:55.155450Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Creating new feature extracting from existing\nfor dataset in combine:\n    dataset['Title'] = dataset.Name.str.extract(' ([A-Za-z]+)\\.', expand=False)\n\npd.crosstab(train_df['Title'], train_df['Sex'])","metadata":{"_uuid":"c916644bd151f3dc8fca900f656d415b4c55e2bc","_cell_guid":"df7f0cd4-992c-4a79-fb19-bf6f0c024d4b","execution":{"iopub.status.busy":"2022-07-28T16:40:55.159246Z","iopub.execute_input":"2022-07-28T16:40:55.160246Z","iopub.status.idle":"2022-07-28T16:40:55.199496Z","shell.execute_reply.started":"2022-07-28T16:40:55.160083Z","shell.execute_reply":"2022-07-28T16:40:55.198745Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#replace many titles with a more common name or classify them as Rare\nfor dataset in combine:\n    dataset['Title'] = dataset['Title'].replace(['Lady', 'Countess','Capt', 'Col',\\\n \t'Don', 'Dr', 'Major', 'Rev', 'Sir', 'Jonkheer', 'Dona'], 'Rare')\n\n    dataset['Title'] = dataset['Title'].replace('Mlle', 'Miss')\n    dataset['Title'] = dataset['Title'].replace('Ms', 'Miss')\n    dataset['Title'] = dataset['Title'].replace('Mme', 'Mrs')\n    \ntrain_df[['Title', 'Survived']].groupby(['Title'], as_index=False).mean()","metadata":{"_uuid":"b8cd938fba61fb4e226c77521b012f4bb8aa01d0","_cell_guid":"553f56d7-002a-ee63-21a4-c0efad10cfe9","execution":{"iopub.status.busy":"2022-07-28T16:40:55.200477Z","iopub.execute_input":"2022-07-28T16:40:55.200846Z","iopub.status.idle":"2022-07-28T16:40:55.230020Z","shell.execute_reply.started":"2022-07-28T16:40:55.200806Z","shell.execute_reply":"2022-07-28T16:40:55.229237Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#convert the categorical titles to ordinal.\ntitle_mapping = {\"Mr\": 1, \"Miss\": 2, \"Mrs\": 3, \"Master\": 4, \"Rare\": 5}\nfor dataset in combine:\n    dataset['Title'] = dataset['Title'].map(title_mapping)\n    dataset['Title'] = dataset['Title'].fillna(0)\n\ntrain_df.head()","metadata":{"_uuid":"e805ad52f0514497b67c3726104ba46d361eb92c","_cell_guid":"67444ebc-4d11-bac1-74a6-059133b6e2e8","execution":{"iopub.status.busy":"2022-07-28T16:40:55.231626Z","iopub.execute_input":"2022-07-28T16:40:55.231914Z","iopub.status.idle":"2022-07-28T16:40:55.269286Z","shell.execute_reply.started":"2022-07-28T16:40:55.231861Z","shell.execute_reply":"2022-07-28T16:40:55.268350Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#drop the Name feature from training and testing datasets. We also do not need the PassengerId feature in the training dataset\ntrain_df = train_df.drop(['Name', 'PassengerId'], axis=1)\ntest_df = test_df.drop(['Name'], axis=1)\ncombine = [train_df, test_df]\ntrain_df.shape, test_df.shape","metadata":{"_uuid":"1da299cf2ffd399fd5b37d74fb40665d16ba5347","_cell_guid":"9d61dded-5ff0-5018-7580-aecb4ea17506","execution":{"iopub.status.busy":"2022-07-28T16:40:55.270922Z","iopub.execute_input":"2022-07-28T16:40:55.271557Z","iopub.status.idle":"2022-07-28T16:40:55.284884Z","shell.execute_reply.started":"2022-07-28T16:40:55.271495Z","shell.execute_reply":"2022-07-28T16:40:55.284043Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Converting categorical feature for Sex:\nfor dataset in combine:\n    dataset['Sex'] = dataset['Sex'].map( {'female': 1, 'male': 0} ).astype(int)\n\ntrain_df.head()","metadata":{"_uuid":"840498eaee7baaca228499b0a5652da9d4edaf37","_cell_guid":"c20c1df2-157c-e5a0-3e24-15a828095c96","execution":{"iopub.status.busy":"2022-07-28T16:40:55.286343Z","iopub.execute_input":"2022-07-28T16:40:55.286923Z","iopub.status.idle":"2022-07-28T16:40:55.319031Z","shell.execute_reply.started":"2022-07-28T16:40:55.286861Z","shell.execute_reply":"2022-07-28T16:40:55.317721Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#guessing missing values is to use other correlated features.\n# grid = sns.FacetGrid(train_df, col='Pclass', hue='Gender')\ngrid = sns.FacetGrid(train_df, row='Pclass', col='Sex', size=2.2, aspect=1.6)\ngrid.map(plt.hist, 'Age', alpha=.5, bins=20)\ngrid.add_legend()","metadata":{"_uuid":"345038c8dd1bac9a9bc5e2cfee13fcc1f833eee0","_cell_guid":"c311c43d-6554-3b52-8ef8-533ca08b2f68","execution":{"iopub.status.busy":"2022-07-28T16:40:55.320299Z","iopub.execute_input":"2022-07-28T16:40:55.320692Z","iopub.status.idle":"2022-07-28T16:40:56.476534Z","shell.execute_reply.started":"2022-07-28T16:40:55.320652Z","shell.execute_reply":"2022-07-28T16:40:56.475746Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Let us start by preparing an empty array to contain guessed Age values based on Pclass x Gender combinations.","metadata":{"_uuid":"6b22ac53d95c7979d5f4580bd5fd29d27155c347","_cell_guid":"a4f166f9-f5f9-1819-66c3-d89dd5b0d8ff"}},{"cell_type":"code","source":"#preparing an empty array to contain guessed Age values based on Pclass x Gender combinations.\nguess_ages = np.zeros((2,3))\nguess_ages","metadata":{"_uuid":"24a0971daa4cbc3aa700bae42e68c17ce9f3a6e2","_cell_guid":"9299523c-dcf1-fb00-e52f-e2fb860a3920","execution":{"iopub.status.busy":"2022-07-28T16:40:56.477913Z","iopub.execute_input":"2022-07-28T16:40:56.478502Z","iopub.status.idle":"2022-07-28T16:40:56.485917Z","shell.execute_reply.started":"2022-07-28T16:40:56.478435Z","shell.execute_reply":"2022-07-28T16:40:56.484972Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#preparing an empty array to contain guessed Age values based on Pclass x Gender combinations.\nfor dataset in combine:\n    for i in range(0, 2):\n        for j in range(0, 3):\n            guess_df = dataset[(dataset['Sex'] == i) & \\\n                                  (dataset['Pclass'] == j+1)]['Age'].dropna()\n\n            # age_mean = guess_df.mean()\n            # age_std = guess_df.std()\n            # age_guess = rnd.uniform(age_mean - age_std, age_mean + age_std)\n\n            age_guess = guess_df.median()\n\n            # Convert random age float to nearest .5 age\n            guess_ages[i,j] = int( age_guess/0.5 + 0.5 ) * 0.5\n            \n    for i in range(0, 2):\n        for j in range(0, 3):\n            dataset.loc[ (dataset.Age.isnull()) & (dataset.Sex == i) & (dataset.Pclass == j+1),\\\n                    'Age'] = guess_ages[i,j]\n\n    dataset['Age'] = dataset['Age'].astype(int)\n\ntrain_df.head()","metadata":{"_uuid":"31198f0ad0dbbb74290ebe135abffa994b8f58f3","_cell_guid":"a4015dfa-a0ab-65bc-0cbe-efecf1eb2569","execution":{"iopub.status.busy":"2022-07-28T16:40:56.487386Z","iopub.execute_input":"2022-07-28T16:40:56.488108Z","iopub.status.idle":"2022-07-28T16:40:56.596052Z","shell.execute_reply.started":"2022-07-28T16:40:56.488047Z","shell.execute_reply":"2022-07-28T16:40:56.595121Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#create Age bands and determine correlations with Survived\ntrain_df['AgeBand'] = pd.cut(train_df['Age'], 5)\ntrain_df[['AgeBand', 'Survived']].groupby(['AgeBand'], as_index=False).mean().sort_values(by='AgeBand', ascending=True)","metadata":{"_uuid":"5c8b4cbb302f439ef0d6278dcfbdafd952675353","_cell_guid":"725d1c84-6323-9d70-5812-baf9994d3aa1","execution":{"iopub.status.busy":"2022-07-28T16:40:56.597436Z","iopub.execute_input":"2022-07-28T16:40:56.597954Z","iopub.status.idle":"2022-07-28T16:40:56.626791Z","shell.execute_reply.started":"2022-07-28T16:40:56.597894Z","shell.execute_reply":"2022-07-28T16:40:56.625950Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#replace Age with ordinals based on these bands.\nfor dataset in combine:    \n    dataset.loc[ dataset['Age'] <= 16, 'Age'] = 0\n    dataset.loc[(dataset['Age'] > 16) & (dataset['Age'] <= 32), 'Age'] = 1\n    dataset.loc[(dataset['Age'] > 32) & (dataset['Age'] <= 48), 'Age'] = 2\n    dataset.loc[(dataset['Age'] > 48) & (dataset['Age'] <= 64), 'Age'] = 3\n    dataset.loc[ dataset['Age'] > 64, 'Age']\ntrain_df.head()","metadata":{"_uuid":"ee13831345f389db407c178f66c19cc8331445b0","_cell_guid":"797b986d-2c45-a9ee-e5b5-088de817c8b2","execution":{"iopub.status.busy":"2022-07-28T16:40:56.628230Z","iopub.execute_input":"2022-07-28T16:40:56.628839Z","iopub.status.idle":"2022-07-28T16:40:56.692560Z","shell.execute_reply.started":"2022-07-28T16:40:56.628778Z","shell.execute_reply":"2022-07-28T16:40:56.691577Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#We can now remove the AgeBand feature.\ntrain_df = train_df.drop(['AgeBand'], axis=1)\ncombine = [train_df, test_df]\ntrain_df.head()","metadata":{"_uuid":"1ea01ccc4a24e8951556d97c990aa0136da19721","_cell_guid":"875e55d4-51b0-5061-b72c-8a23946133a3","execution":{"iopub.status.busy":"2022-07-28T16:40:56.694020Z","iopub.execute_input":"2022-07-28T16:40:56.694655Z","iopub.status.idle":"2022-07-28T16:40:56.721020Z","shell.execute_reply.started":"2022-07-28T16:40:56.694585Z","shell.execute_reply":"2022-07-28T16:40:56.720263Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Create a new feature for FamilySize which combines Parch and SibSp. This will enable us to drop Parch and SibSp from our datasets.\nfor dataset in combine:\n    dataset['FamilySize'] = dataset['SibSp'] + dataset['Parch'] + 1\n\ntrain_df[['FamilySize', 'Survived']].groupby(['FamilySize'], as_index=False).mean().sort_values(by='Survived', ascending=False)","metadata":{"_uuid":"33d1236ce4a8ab888b9fac2d5af1c78d174b32c7","_cell_guid":"7e6c04ed-cfaa-3139-4378-574fd095d6ba","execution":{"iopub.status.busy":"2022-07-28T16:40:56.722445Z","iopub.execute_input":"2022-07-28T16:40:56.722761Z","iopub.status.idle":"2022-07-28T16:40:56.751247Z","shell.execute_reply.started":"2022-07-28T16:40:56.722707Z","shell.execute_reply":"2022-07-28T16:40:56.750415Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#create another feature called IsAlone\nfor dataset in combine:\n    dataset['IsAlone'] = 0\n    dataset.loc[dataset['FamilySize'] == 1, 'IsAlone'] = 1\n\ntrain_df[['IsAlone', 'Survived']].groupby(['IsAlone'], as_index=False).mean()","metadata":{"_uuid":"3b8db81cc3513b088c6bcd9cd1938156fe77992f","_cell_guid":"5c778c69-a9ae-1b6b-44fe-a0898d07be7a","execution":{"iopub.status.busy":"2022-07-28T16:40:56.752480Z","iopub.execute_input":"2022-07-28T16:40:56.752943Z","iopub.status.idle":"2022-07-28T16:40:56.780846Z","shell.execute_reply.started":"2022-07-28T16:40:56.752876Z","shell.execute_reply":"2022-07-28T16:40:56.779888Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#drop Parch, SibSp, and FamilySize features in favor of IsAlone.\ntrain_df = train_df.drop(['Parch', 'SibSp', 'FamilySize'], axis=1)\ntest_df = test_df.drop(['Parch', 'SibSp', 'FamilySize'], axis=1)\ncombine = [train_df, test_df]\n\ntrain_df.head()","metadata":{"_uuid":"1e3479690ef7cd8ee10538d4f39d7117246887f0","_cell_guid":"74ee56a6-7357-f3bc-b605-6c41f8aa6566","execution":{"iopub.status.busy":"2022-07-28T16:40:56.782457Z","iopub.execute_input":"2022-07-28T16:40:56.783088Z","iopub.status.idle":"2022-07-28T16:40:56.810027Z","shell.execute_reply.started":"2022-07-28T16:40:56.783027Z","shell.execute_reply":"2022-07-28T16:40:56.808967Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#create an artificial feature combining Pclass and Age\nfor dataset in combine:\n    dataset['Age*Class'] = dataset.Age * dataset.Pclass\n\ntrain_df.loc[:, ['Age*Class', 'Age', 'Pclass']].head(10)","metadata":{"_uuid":"aac2c5340c06210a8b0199e15461e9049fbf2cff","_cell_guid":"305402aa-1ea1-c245-c367-056eef8fe453","execution":{"iopub.status.busy":"2022-07-28T16:40:56.811428Z","iopub.execute_input":"2022-07-28T16:40:56.811698Z","iopub.status.idle":"2022-07-28T16:40:56.831469Z","shell.execute_reply.started":"2022-07-28T16:40:56.811652Z","shell.execute_reply":"2022-07-28T16:40:56.830619Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Embarked feature takes S, Q, C values based on port of embarkation. Our training dataset has two missing values. We simply fill these with the most common occurance.\nfreq_port = train_df.Embarked.dropna().mode()[0]\nfreq_port","metadata":{"_uuid":"1e3f8af166f60a1b3125a6b046eff5fff02d63cf","_cell_guid":"bf351113-9b7f-ef56-7211-e8dd00665b18","execution":{"iopub.status.busy":"2022-07-28T16:40:56.832742Z","iopub.execute_input":"2022-07-28T16:40:56.833266Z","iopub.status.idle":"2022-07-28T16:40:56.848746Z","shell.execute_reply.started":"2022-07-28T16:40:56.833219Z","shell.execute_reply":"2022-07-28T16:40:56.847604Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for dataset in combine:\n    dataset['Embarked'] = dataset['Embarked'].fillna(freq_port)\n    \ntrain_df[['Embarked', 'Survived']].groupby(['Embarked'], as_index=False).mean().sort_values(by='Survived', ascending=False)","metadata":{"_uuid":"d85b5575fb45f25749298641f6a0a38803e1ff22","_cell_guid":"51c21fcc-f066-cd80-18c8-3d140be6cbae","execution":{"iopub.status.busy":"2022-07-28T16:40:56.849897Z","iopub.execute_input":"2022-07-28T16:40:56.850315Z","iopub.status.idle":"2022-07-28T16:40:56.874308Z","shell.execute_reply.started":"2022-07-28T16:40:56.850273Z","shell.execute_reply":"2022-07-28T16:40:56.873504Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Convert the EmbarkedFill feature by creating a new numeric Port feature\nfor dataset in combine:\n    dataset['Embarked'] = dataset['Embarked'].map( {'S': 0, 'C': 1, 'Q': 2} ).astype(int)\n\ntrain_df.head()","metadata":{"_uuid":"e480a1ef145de0b023821134896391d568a6f4f9","_cell_guid":"89a91d76-2cc0-9bbb-c5c5-3c9ecae33c66","execution":{"iopub.status.busy":"2022-07-28T16:40:56.875455Z","iopub.execute_input":"2022-07-28T16:40:56.875854Z","iopub.status.idle":"2022-07-28T16:40:56.906208Z","shell.execute_reply.started":"2022-07-28T16:40:56.875799Z","shell.execute_reply":"2022-07-28T16:40:56.905245Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Round off the fare to two decimals as it represents currency\ntest_df['Fare'].fillna(test_df['Fare'].dropna().median(), inplace=True)\ntest_df.head()","metadata":{"_uuid":"aacb62f3526072a84795a178bd59222378bab180","_cell_guid":"3600cb86-cf5f-d87b-1b33-638dc8db1564","execution":{"iopub.status.busy":"2022-07-28T16:40:56.907851Z","iopub.execute_input":"2022-07-28T16:40:56.908550Z","iopub.status.idle":"2022-07-28T16:40:56.935756Z","shell.execute_reply.started":"2022-07-28T16:40:56.908486Z","shell.execute_reply":"2022-07-28T16:40:56.935027Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#We can now create FareBand.\ntrain_df['FareBand'] = pd.qcut(train_df['Fare'], 4)\ntrain_df[['FareBand', 'Survived']].groupby(['FareBand'], as_index=False).mean().sort_values(by='FareBand', ascending=True)","metadata":{"_uuid":"b9a78f6b4c72520d4ad99d2c89c84c591216098d","_cell_guid":"0e9018b1-ced5-9999-8ce1-258a0952cbf2","execution":{"iopub.status.busy":"2022-07-28T16:40:56.937031Z","iopub.execute_input":"2022-07-28T16:40:56.937508Z","iopub.status.idle":"2022-07-28T16:40:56.968690Z","shell.execute_reply.started":"2022-07-28T16:40:56.937451Z","shell.execute_reply":"2022-07-28T16:40:56.967703Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Convert the Fare feature to ordinal values based on the FareBand\nfor dataset in combine:\n    dataset.loc[ dataset['Fare'] <= 7.91, 'Fare'] = 0\n    dataset.loc[(dataset['Fare'] > 7.91) & (dataset['Fare'] <= 14.454), 'Fare'] = 1\n    dataset.loc[(dataset['Fare'] > 14.454) & (dataset['Fare'] <= 31), 'Fare']   = 2\n    dataset.loc[ dataset['Fare'] > 31, 'Fare'] = 3\n    dataset['Fare'] = dataset['Fare'].astype(int)\n\ntrain_df = train_df.drop(['FareBand'], axis=1)\ncombine = [train_df, test_df]\n    \ntrain_df.head(10)","metadata":{"_uuid":"640f305061ec4221a45ba250f8d54bb391035a57","_cell_guid":"385f217a-4e00-76dc-1570-1de4eec0c29c","execution":{"iopub.status.busy":"2022-07-28T16:40:56.970119Z","iopub.execute_input":"2022-07-28T16:40:56.970439Z","iopub.status.idle":"2022-07-28T16:40:57.034982Z","shell.execute_reply.started":"2022-07-28T16:40:56.970383Z","shell.execute_reply":"2022-07-28T16:40:57.034259Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Exam the test dataset.\ntest_df.head(10)","metadata":{"_uuid":"8453cecad81fcc44de3f4e4e4c3ce6afa977740d","_cell_guid":"d2334d33-4fe5-964d-beac-6aa620066e15","execution":{"iopub.status.busy":"2022-07-28T16:41:49.866813Z","iopub.execute_input":"2022-07-28T16:41:49.867133Z","iopub.status.idle":"2022-07-28T16:41:49.892991Z","shell.execute_reply.started":"2022-07-28T16:41:49.867075Z","shell.execute_reply":"2022-07-28T16:41:49.892009Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#To identify relationship between output (Survived or not) with other variables or features (Gender, Age, Port...) and train our model with a given dataset:\n#Logistic Regression\n#KNN or k-Nearest Neighbors\n#Support Vector Machines\n#Naive Bayes classifier\n#Decision Tree\n#Random Forrest\n#Perceptron\n#Artificial neural network\n#RVM or Relevance Vector Machine\n\nX_train = train_df.drop(\"Survived\", axis=1)\nY_train = train_df[\"Survived\"]\nX_test  = test_df.drop(\"PassengerId\", axis=1).copy()\nX_train.shape, Y_train.shape, X_test.shape","metadata":{"_uuid":"04d2235855f40cffd81f76b977a500fceaae87ad","_cell_guid":"0acf54f9-6cf5-24b5-72d9-29b30052823a","execution":{"iopub.status.busy":"2022-07-28T17:12:49.935921Z","iopub.execute_input":"2022-07-28T17:12:49.936274Z","iopub.status.idle":"2022-07-28T17:12:49.948071Z","shell.execute_reply.started":"2022-07-28T17:12:49.936222Z","shell.execute_reply":"2022-07-28T17:12:49.947028Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Logistic Regression\n\nlogreg = LogisticRegression()\nlogreg.fit(X_train, Y_train)\nY_pred = logreg.predict(X_test)\nacc_log = round(logreg.score(X_train, Y_train) * 100, 2)\nacc_log\n\ncoeff_df = pd.DataFrame(train_df.columns.delete(0))\ncoeff_df.columns = ['Feature']\ncoeff_df[\"Correlation\"] = pd.Series(logreg.coef_[0])\n\ncoeff_df.sort_values(by='Correlation', ascending=False)","metadata":{"_uuid":"a649b9c53f4c7b40694f60f5c8dc14ec5ef519ec","_cell_guid":"0edd9322-db0b-9c37-172d-a3a4f8dec229","execution":{"iopub.status.busy":"2022-07-28T17:14:45.043203Z","iopub.execute_input":"2022-07-28T17:14:45.043524Z","iopub.status.idle":"2022-07-28T17:14:45.071671Z","shell.execute_reply.started":"2022-07-28T17:14:45.043473Z","shell.execute_reply":"2022-07-28T17:14:45.070837Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Support Vector Machines\n\nsvc = SVC()\nsvc.fit(X_train, Y_train)\nY_pred = svc.predict(X_test)\nacc_svc = round(svc.score(X_train, Y_train) * 100, 2)\nacc_svc","metadata":{"_uuid":"60039d5377da49f1aa9ac4a924331328bd69add1","_cell_guid":"7a63bf04-a410-9c81-5310-bdef7963298f","execution":{"iopub.status.busy":"2022-07-28T17:15:53.672260Z","iopub.execute_input":"2022-07-28T17:15:53.672576Z","iopub.status.idle":"2022-07-28T17:15:53.728299Z","shell.execute_reply.started":"2022-07-28T17:15:53.672535Z","shell.execute_reply":"2022-07-28T17:15:53.727475Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#KNN confidence score is better than Logistics Regression but worse than SVM\nknn = KNeighborsClassifier(n_neighbors = 3)\nknn.fit(X_train, Y_train)\nY_pred = knn.predict(X_test)\nacc_knn = round(knn.score(X_train, Y_train) * 100, 2)\nacc_knn","metadata":{"_uuid":"54d86cd45703d459d452f89572771deaa8877999","_cell_guid":"ca14ae53-f05e-eb73-201c-064d7c3ed610","execution":{"iopub.status.busy":"2022-07-28T17:16:30.022007Z","iopub.execute_input":"2022-07-28T17:16:30.022656Z","iopub.status.idle":"2022-07-28T17:16:30.045553Z","shell.execute_reply.started":"2022-07-28T17:16:30.022602Z","shell.execute_reply":"2022-07-28T17:16:30.044691Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Gaussian Naive Bayes\n\ngaussian = GaussianNB()\ngaussian.fit(X_train, Y_train)\nY_pred = gaussian.predict(X_test)\nacc_gaussian = round(gaussian.score(X_train, Y_train) * 100, 2)\nacc_gaussian","metadata":{"_uuid":"723c835c29e8727bc9bad4b564731f2ca98025d0","_cell_guid":"50378071-7043-ed8d-a782-70c947520dae","execution":{"iopub.status.busy":"2022-07-28T17:16:46.767710Z","iopub.execute_input":"2022-07-28T17:16:46.768060Z","iopub.status.idle":"2022-07-28T17:16:46.781580Z","shell.execute_reply.started":"2022-07-28T17:16:46.767999Z","shell.execute_reply":"2022-07-28T17:16:46.780581Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Perceptron\n\nperceptron = Perceptron()\nperceptron.fit(X_train, Y_train)\nY_pred = perceptron.predict(X_test)\nacc_perceptron = round(perceptron.score(X_train, Y_train) * 100, 2)\nacc_perceptron","metadata":{"_uuid":"c19d08949f9c3a26931e28adedc848b4deaa8ab6","_cell_guid":"ccc22a86-b7cb-c2dd-74bd-53b218d6ed0d","execution":{"iopub.status.busy":"2022-07-28T17:17:36.457283Z","iopub.execute_input":"2022-07-28T17:17:36.457802Z","iopub.status.idle":"2022-07-28T17:17:36.472498Z","shell.execute_reply.started":"2022-07-28T17:17:36.457758Z","shell.execute_reply":"2022-07-28T17:17:36.470083Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Linear SVC\n\nlinear_svc = LinearSVC()\nlinear_svc.fit(X_train, Y_train)\nY_pred = linear_svc.predict(X_test)\nacc_linear_svc = round(linear_svc.score(X_train, Y_train) * 100, 2)\nacc_linear_svc","metadata":{"_uuid":"52ea4f44dd626448dd2199cb284b592670b1394b","_cell_guid":"a4d56857-9432-55bb-14c0-52ebeb64d198","execution":{"iopub.status.busy":"2022-07-28T17:17:46.011594Z","iopub.execute_input":"2022-07-28T17:17:46.012100Z","iopub.status.idle":"2022-07-28T17:17:46.075701Z","shell.execute_reply.started":"2022-07-28T17:17:46.012051Z","shell.execute_reply":"2022-07-28T17:17:46.074790Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Stochastic Gradient Descent\n\nsgd = SGDClassifier()\nsgd.fit(X_train, Y_train)\nY_pred = sgd.predict(X_test)\nacc_sgd = round(sgd.score(X_train, Y_train) * 100, 2)\nacc_sgd","metadata":{"_uuid":"3a016c1f24da59c85648204302d61ea15920e740","_cell_guid":"dc98ed72-3aeb-861f-804d-b6e3d178bf4b","execution":{"iopub.status.busy":"2022-07-28T17:17:51.119127Z","iopub.execute_input":"2022-07-28T17:17:51.119570Z","iopub.status.idle":"2022-07-28T17:17:51.134603Z","shell.execute_reply.started":"2022-07-28T17:17:51.119509Z","shell.execute_reply":"2022-07-28T17:17:51.133491Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Decision Tree\nimport time\nstart = time.time()\n\ndecision_tree = DecisionTreeClassifier()\ndecision_tree.fit(X_train, Y_train)\nY_pred = decision_tree.predict(X_test)\nacc_decision_tree = round(decision_tree.score(X_train, Y_train) * 100, 2)\nacc_decision_tree\n\nprint(f'Decision Tree runtime elapsed: {round(time.time() - start, 2)} seconds.')","metadata":{"_uuid":"1f94308b23b934123c03067e84027b507b989e52","_cell_guid":"dd85f2b7-ace2-0306-b4ec-79c68cd3fea0","execution":{"iopub.status.busy":"2022-07-28T18:18:05.929917Z","iopub.execute_input":"2022-07-28T18:18:05.930308Z","iopub.status.idle":"2022-07-28T18:18:05.942545Z","shell.execute_reply.started":"2022-07-28T18:18:05.930251Z","shell.execute_reply":"2022-07-28T18:18:05.941308Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Random Forest\n\nstart = time.time()\n\nrandom_forest = RandomForestClassifier(n_estimators=100)\nrandom_forest.fit(X_train, Y_train)\nY_pred = random_forest.predict(X_test)\nrandom_forest.score(X_train, Y_train)\nacc_random_forest = round(random_forest.score(X_train, Y_train) * 100, 2)\nacc_random_forest\n\nprint(f'Decision Tree runtime elapsed: {round(time.time() - start, 2)} seconds.')","metadata":{"_uuid":"483c647d2759a2703d20785a44f51b6dee47d0db","_cell_guid":"f0694a8e-b618-8ed9-6f0d-8c6fba2c4567","execution":{"iopub.status.busy":"2022-07-28T18:18:37.738001Z","iopub.execute_input":"2022-07-28T18:18:37.738497Z","iopub.status.idle":"2022-07-28T18:18:37.913023Z","shell.execute_reply.started":"2022-07-28T18:18:37.738452Z","shell.execute_reply":"2022-07-28T18:18:37.912165Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Model Evaluation:\nmodels = pd.DataFrame({\n    'Model': ['Support Vector Machines', 'KNN', 'Logistic Regression', \n              'Random Forest', 'Naive Bayes', 'Perceptron', \n              'Stochastic Gradient Decent', 'Linear SVC', \n              'Decision Tree'],\n    'Score': [acc_svc, acc_knn, acc_log, \n              acc_random_forest, acc_gaussian, acc_perceptron, \n              acc_sgd, acc_linear_svc, acc_decision_tree]})\nmodels.sort_values(by='Score', ascending=False)","metadata":{"_uuid":"06a52babe50e0dd837b553c78fc73872168e1c7d","_cell_guid":"1f3cebe0-31af-70b2-1ce4-0fd406bcdfc6","execution":{"iopub.status.busy":"2022-07-28T17:19:03.626420Z","iopub.execute_input":"2022-07-28T17:19:03.626735Z","iopub.status.idle":"2022-07-28T17:19:03.647440Z","shell.execute_reply.started":"2022-07-28T17:19:03.626688Z","shell.execute_reply":"2022-07-28T17:19:03.646672Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Runtime Comparison: \n#Random Forest take 17 times longer to run than the Decision Tree for this data set.\n\n#submission = pd.DataFrame({\n #       \"PassengerId\": test_df[\"PassengerId\"],\n  #      \"Survived\": Y_pred\n   # })\n# submission.to_csv('../output/submission.csv', index=False)","metadata":{"_uuid":"82b31ea933b3026bd038a8370d651efdcdb3e4d7","_cell_guid":"28854d36-051f-3ef0-5535-fa5ba6a9bef7","execution":{"iopub.status.busy":"2022-07-28T17:19:24.970737Z","iopub.execute_input":"2022-07-28T17:19:24.971094Z","iopub.status.idle":"2022-07-28T17:19:24.976625Z","shell.execute_reply.started":"2022-07-28T17:19:24.971028Z","shell.execute_reply":"2022-07-28T17:19:24.975890Z"},"trusted":true},"execution_count":null,"outputs":[]}]}