{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np \nimport pandas as pd \nimport matplotlib\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport scipy\nimport patsy\nimport sklearn\nimport os\nimport sys\n\nfrom tabulate import tabulate\n\n%matplotlib inline","metadata":{"execution":{"iopub.status.busy":"2022-08-04T20:21:08.839350Z","iopub.execute_input":"2022-08-04T20:21:08.839959Z","iopub.status.idle":"2022-08-04T20:21:09.849527Z","shell.execute_reply.started":"2022-08-04T20:21:08.839897Z","shell.execute_reply":"2022-08-04T20:21:09.848521Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# get helper file\nshared_path = '../usr/lib'\nsys.path.append(shared_path)\n\nimport helper_1 as helper","metadata":{"execution":{"iopub.status.busy":"2022-08-04T20:21:09.851866Z","iopub.execute_input":"2022-08-04T20:21:09.852305Z","iopub.status.idle":"2022-08-04T20:21:09.866234Z","shell.execute_reply.started":"2022-08-04T20:21:09.852262Z","shell.execute_reply":"2022-08-04T20:21:09.865222Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Import Data","metadata":{}},{"cell_type":"code","source":"# Kaggle\ntrain_data = pd.read_csv(\"../input/titanic/train.csv\")\ntest_data = pd.read_csv(\"../input/titanic/test.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-08-04T20:21:09.867534Z","iopub.execute_input":"2022-08-04T20:21:09.867818Z","iopub.status.idle":"2022-08-04T20:21:09.896450Z","shell.execute_reply.started":"2022-08-04T20:21:09.867790Z","shell.execute_reply":"2022-08-04T20:21:09.895514Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Local Jupyter\ndf = train_data\norig = df.copy()","metadata":{"execution":{"iopub.status.busy":"2022-08-04T20:21:09.897741Z","iopub.execute_input":"2022-08-04T20:21:09.898051Z","iopub.status.idle":"2022-08-04T20:21:09.903428Z","shell.execute_reply.started":"2022-08-04T20:21:09.898019Z","shell.execute_reply":"2022-08-04T20:21:09.902204Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":" Description\n* pclass: Passenger class (1 = 1st; 2 = 2nd; 3 = 3rd)\n* survival: A Boolean indicating whether the passenger survived or not (0 = No; 1 = Yes); this is our target\n* name: A field rich in information as it contains title and family names\n* sex: male/female\n* age: Age, asignificant portion of values aremissing\n* sibsp: Number of siblings/spouses aboard\n* parch: Number of parents/children aboard\n* ticket: Ticket number.\n* fare: Passenger fare (British Pound).\n* cabin: Doesthe location of the cabin influence chances of survival?\n* embarked: Port of embarkation (C = Cherbourg; Q = Queenstown; S = Southampton)\n* boat: Lifeboat, many missing values\n* body: Body Identification Number","metadata":{}},{"cell_type":"markdown","source":"# EDA","metadata":{}},{"cell_type":"code","source":"df.info()","metadata":{"execution":{"iopub.status.busy":"2022-08-04T20:21:09.906098Z","iopub.execute_input":"2022-08-04T20:21:09.906588Z","iopub.status.idle":"2022-08-04T20:21:09.931963Z","shell.execute_reply.started":"2022-08-04T20:21:09.906556Z","shell.execute_reply":"2022-08-04T20:21:09.931218Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.describe()","metadata":{"execution":{"iopub.status.busy":"2022-08-04T20:21:09.933312Z","iopub.execute_input":"2022-08-04T20:21:09.933739Z","iopub.status.idle":"2022-08-04T20:21:09.972597Z","shell.execute_reply.started":"2022-08-04T20:21:09.933706Z","shell.execute_reply":"2022-08-04T20:21:09.971867Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.dtypes","metadata":{"execution":{"iopub.status.busy":"2022-08-04T20:21:09.973600Z","iopub.execute_input":"2022-08-04T20:21:09.974011Z","iopub.status.idle":"2022-08-04T20:21:09.981832Z","shell.execute_reply.started":"2022-08-04T20:21:09.973981Z","shell.execute_reply":"2022-08-04T20:21:09.980555Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-04T20:21:10.126028Z","iopub.execute_input":"2022-08-04T20:21:10.126675Z","iopub.status.idle":"2022-08-04T20:21:10.144505Z","shell.execute_reply.started":"2022-08-04T20:21:10.126614Z","shell.execute_reply":"2022-08-04T20:21:10.143685Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.tail()","metadata":{"execution":{"iopub.status.busy":"2022-08-04T20:21:10.314625Z","iopub.execute_input":"2022-08-04T20:21:10.315169Z","iopub.status.idle":"2022-08-04T20:21:10.332187Z","shell.execute_reply.started":"2022-08-04T20:21:10.315135Z","shell.execute_reply":"2022-08-04T20:21:10.330990Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.isna().any()","metadata":{"execution":{"iopub.status.busy":"2022-08-04T20:21:10.489079Z","iopub.execute_input":"2022-08-04T20:21:10.489479Z","iopub.status.idle":"2022-08-04T20:21:10.498152Z","shell.execute_reply.started":"2022-08-04T20:21:10.489444Z","shell.execute_reply":"2022-08-04T20:21:10.497358Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Data Cleansing I","metadata":{}},{"cell_type":"markdown","source":"### Unused Columns","metadata":{}},{"cell_type":"markdown","source":"Name passengerid and ticket will probably not be useful and have very little impact on our survived prediction so lets drop them","metadata":{}},{"cell_type":"code","source":"df.drop(['Name', 'PassengerId', 'Ticket'], axis=1, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T20:21:10.662457Z","iopub.execute_input":"2022-08-04T20:21:10.663124Z","iopub.status.idle":"2022-08-04T20:21:10.671736Z","shell.execute_reply.started":"2022-08-04T20:21:10.663074Z","shell.execute_reply":"2022-08-04T20:21:10.670597Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.columns","metadata":{"execution":{"iopub.status.busy":"2022-08-04T20:21:10.841206Z","iopub.execute_input":"2022-08-04T20:21:10.841570Z","iopub.status.idle":"2022-08-04T20:21:10.848208Z","shell.execute_reply.started":"2022-08-04T20:21:10.841540Z","shell.execute_reply":"2022-08-04T20:21:10.847139Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Missing Values","metadata":{}},{"cell_type":"code","source":"df.Age.isnull().value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-08-04T20:21:11.013507Z","iopub.execute_input":"2022-08-04T20:21:11.014236Z","iopub.status.idle":"2022-08-04T20:21:11.025508Z","shell.execute_reply.started":"2022-08-04T20:21:11.014186Z","shell.execute_reply":"2022-08-04T20:21:11.024545Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.Cabin.isnull().value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-08-04T20:21:11.188409Z","iopub.execute_input":"2022-08-04T20:21:11.188795Z","iopub.status.idle":"2022-08-04T20:21:11.197865Z","shell.execute_reply.started":"2022-08-04T20:21:11.188762Z","shell.execute_reply":"2022-08-04T20:21:11.197134Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.Embarked.isnull().value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-08-04T20:21:11.362131Z","iopub.execute_input":"2022-08-04T20:21:11.362664Z","iopub.status.idle":"2022-08-04T20:21:11.371921Z","shell.execute_reply.started":"2022-08-04T20:21:11.362631Z","shell.execute_reply":"2022-08-04T20:21:11.370967Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Age","metadata":{}},{"cell_type":"markdown","source":"We need to fill 177 values with imputed values","metadata":{}},{"cell_type":"code","source":"fig, (ax1, ax2, ax3) = plt.subplots(1,3, figsize=(20,5))\n\n\nax1.hist(df.Age, density=True)\nax2.hist(df.Age, bins=20, density=True)\nax3.hist(df.Age, bins=40, density=True)\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-04T20:21:11.536178Z","iopub.execute_input":"2022-08-04T20:21:11.536727Z","iopub.status.idle":"2022-08-04T20:21:12.110197Z","shell.execute_reply.started":"2022-08-04T20:21:11.536691Z","shell.execute_reply":"2022-08-04T20:21:12.109223Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"This is depending on the bin size so lets to a kernel density estimation to see if it will smooth it out","metadata":{}},{"cell_type":"code","source":"df['Age'].plot(kind='kde')","metadata":{"execution":{"iopub.status.busy":"2022-08-04T20:21:12.112605Z","iopub.execute_input":"2022-08-04T20:21:12.112928Z","iopub.status.idle":"2022-08-04T20:21:12.335491Z","shell.execute_reply.started":"2022-08-04T20:21:12.112896Z","shell.execute_reply":"2022-08-04T20:21:12.334556Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"This looks kind of like a normal distribution but also has a long tail so seems more like a gamma, lets try to model it with both","metadata":{}},{"cell_type":"code","source":"import scipy.stats as stats\n\n# get parameters for normal dist mean and std. Scipy uses std as Scale parameter\nage_mean = df.Age.mean()\nage_std = df.Age.std()\n\n# get parameters for gamma dist k=shape, theta=scale using Method of Moments\nk = (df.Age.mean() ** 2) / df.Age.var()\nth = df.Age.var() / df.Age.mean()\n\n# use those as parameters for the normal distribution, note scipy uses std as parameter for norm but for gamma using MoM we use var\nnorm_dist = stats.norm(age_mean, age_std)\ngamm_dist = stats.gamma(k,loc=0, scale=th)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T20:21:12.336918Z","iopub.execute_input":"2022-08-04T20:21:12.337231Z","iopub.status.idle":"2022-08-04T20:21:12.347225Z","shell.execute_reply.started":"2022-08-04T20:21:12.337199Z","shell.execute_reply":"2022-08-04T20:21:12.346325Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"values = [i for i in range(0,100)]\nprobs = [norm_dist.pdf(value) for value in values]\n\nfig, ax = plt.subplots()\n\nax.plot(values, norm_dist.pdf(values), label='normal')\nax.plot(values, gamm_dist.pdf(values), label='gamma')\n\nax.hist(df.Age, density=True, bins=20)\n\nplt.legend()\nplt.show()\nplt.close()","metadata":{"execution":{"iopub.status.busy":"2022-08-04T20:21:12.348690Z","iopub.execute_input":"2022-08-04T20:21:12.349017Z","iopub.status.idle":"2022-08-04T20:21:12.660675Z","shell.execute_reply.started":"2022-08-04T20:21:12.348985Z","shell.execute_reply":"2022-08-04T20:21:12.659964Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"This looks like gamma will be a better model for this data","metadata":{}},{"cell_type":"markdown","source":"using MoM to estimate our parameters for a gamma distribution, we have parameters k and $\\theta$\n\nsrc:(https://en.wikipedia.org/wiki/Gamma_distribution)","metadata":{}},{"cell_type":"code","source":"# k = (df1.Age.mean() ** 2) / df1.Age.var()\n# th = df1.Age.var() / df1.Age.mean()\n\nprint(k)\nprint(th)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T20:21:12.662223Z","iopub.execute_input":"2022-08-04T20:21:12.662622Z","iopub.status.idle":"2022-08-04T20:21:12.667742Z","shell.execute_reply.started":"2022-08-04T20:21:12.662585Z","shell.execute_reply":"2022-08-04T20:21:12.666795Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from scipy.stats import gamma\nfrom scipy.stats import norm\n\ndef add_gammas(data, col):\n    # get parameters for gamma\n    k = (data[col].mean() ** 2) / data[col].var()\n    th = data[col].var() / data[col].mean()\n    \n    # get number of nulls\n    _vals = data[col].isna().value_counts()[True]\n    # get index of nulls\n    _idx = data[data[col].isnull()].index\n    \n    return pd.Series(gamma.rvs(k, loc=0, scale=th, size=_vals), index=_idx)\n    ","metadata":{"execution":{"iopub.status.busy":"2022-08-04T20:21:12.669590Z","iopub.execute_input":"2022-08-04T20:21:12.670011Z","iopub.status.idle":"2022-08-04T20:21:12.678627Z","shell.execute_reply.started":"2022-08-04T20:21:12.669971Z","shell.execute_reply":"2022-08-04T20:21:12.677778Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# unit test for function\narr = add_gammas(df, 'Age')\narr","metadata":{"execution":{"iopub.status.busy":"2022-08-04T20:21:12.679863Z","iopub.execute_input":"2022-08-04T20:21:12.680330Z","iopub.status.idle":"2022-08-04T20:21:12.705740Z","shell.execute_reply.started":"2022-08-04T20:21:12.680298Z","shell.execute_reply":"2022-08-04T20:21:12.704609Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test = df[['Age']].copy()\n# impute values\ndf_test['mean'] = df['Age'].fillna(np.mean(df.Age))\ndf_test['median'] = df['Age'].fillna(np.median(df.Age)) # cannot do median since there are missing values, needs all values before you can calc median\ndf_test['gamma'] = df['Age'].fillna(add_gammas(df, 'Age'))\n\n\ndf_test[df_test.Age.isnull()][:10]","metadata":{"execution":{"iopub.status.busy":"2022-08-04T20:21:12.845924Z","iopub.execute_input":"2022-08-04T20:21:12.846619Z","iopub.status.idle":"2022-08-04T20:21:12.870730Z","shell.execute_reply.started":"2022-08-04T20:21:12.846565Z","shell.execute_reply":"2022-08-04T20:21:12.869685Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test.describe()","metadata":{"execution":{"iopub.status.busy":"2022-08-04T20:21:13.062499Z","iopub.execute_input":"2022-08-04T20:21:13.062851Z","iopub.status.idle":"2022-08-04T20:21:13.090450Z","shell.execute_reply.started":"2022-08-04T20:21:13.062820Z","shell.execute_reply":"2022-08-04T20:21:13.089609Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"mean, std, and bounds look pretty representative for both gamma and the mean imputed values. Gamma seems to be a little more accurate with the bounds","metadata":{}},{"cell_type":"code","source":"fig, ax = plt.subplots(figsize=(12,7))\n\nbinsize=20\n\nax.hist(df_test['Age'], color='b',alpha=0.6, label='age', bins=binsize)\nax.hist(df_test['gamma'], color='orange', alpha=0.6,label='gamma', bins=binsize)\nax.hist(df_test['mean'], color='g', alpha=0.5,label='mean', bins=binsize)\n\nplt.legend()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-04T20:21:13.279071Z","iopub.execute_input":"2022-08-04T20:21:13.279493Z","iopub.status.idle":"2022-08-04T20:21:13.586130Z","shell.execute_reply.started":"2022-08-04T20:21:13.279456Z","shell.execute_reply":"2022-08-04T20:21:13.585249Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Here we see that Age and mean graphs overlap exactly except for about the mean, which is where all the imputed values went. Gamma spread them about across the distribution.","metadata":{}},{"cell_type":"code","source":"df_test.plot(kind='kde', figsize=(12,7))","metadata":{"execution":{"iopub.status.busy":"2022-08-04T20:21:13.588440Z","iopub.execute_input":"2022-08-04T20:21:13.588732Z","iopub.status.idle":"2022-08-04T20:21:13.905373Z","shell.execute_reply.started":"2022-08-04T20:21:13.588704Z","shell.execute_reply":"2022-08-04T20:21:13.904459Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"This looks representative so we should be able to sample from this distribution and impute our missing age values. We also see the mean imputation technique as having a much higher proportion of values near the mean which of course is what we are doing and it had a slight impact on the bound and std. Sampling from our gamma distribution looks like a better technique so we will go with that.","metadata":{}},{"cell_type":"code","source":"df['Age_Gamma'] = df_test.gamma","metadata":{"execution":{"iopub.status.busy":"2022-08-04T20:21:13.907515Z","iopub.execute_input":"2022-08-04T20:21:13.907829Z","iopub.status.idle":"2022-08-04T20:21:13.912518Z","shell.execute_reply.started":"2022-08-04T20:21:13.907796Z","shell.execute_reply":"2022-08-04T20:21:13.911692Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.drop('Age', axis=1, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T20:21:13.914054Z","iopub.execute_input":"2022-08-04T20:21:13.914383Z","iopub.status.idle":"2022-08-04T20:21:13.925900Z","shell.execute_reply.started":"2022-08-04T20:21:13.914352Z","shell.execute_reply":"2022-08-04T20:21:13.925058Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.isna().any()","metadata":{"execution":{"iopub.status.busy":"2022-08-04T20:21:14.162856Z","iopub.execute_input":"2022-08-04T20:21:14.163222Z","iopub.status.idle":"2022-08-04T20:21:14.174167Z","shell.execute_reply.started":"2022-08-04T20:21:14.163191Z","shell.execute_reply":"2022-08-04T20:21:14.173418Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Embarked","metadata":{}},{"cell_type":"code","source":"df['Embarked'].hist()","metadata":{"execution":{"iopub.status.busy":"2022-08-04T20:21:14.336134Z","iopub.execute_input":"2022-08-04T20:21:14.336646Z","iopub.status.idle":"2022-08-04T20:21:14.459204Z","shell.execute_reply.started":"2022-08-04T20:21:14.336611Z","shell.execute_reply":"2022-08-04T20:21:14.458458Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['Embarked'].isna().value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-08-04T20:21:14.504208Z","iopub.execute_input":"2022-08-04T20:21:14.504743Z","iopub.status.idle":"2022-08-04T20:21:14.512391Z","shell.execute_reply.started":"2022-08-04T20:21:14.504707Z","shell.execute_reply":"2022-08-04T20:21:14.511587Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# only two here are empty so just fill with most frequent\ndf['Embarked'].fillna('S', inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T20:21:14.669902Z","iopub.execute_input":"2022-08-04T20:21:14.670491Z","iopub.status.idle":"2022-08-04T20:21:14.675834Z","shell.execute_reply.started":"2022-08-04T20:21:14.670435Z","shell.execute_reply":"2022-08-04T20:21:14.674395Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Cabin","metadata":{}},{"cell_type":"code","source":"df['Cabin'].isna().value_counts()","metadata":{"tags":[],"execution":{"iopub.status.busy":"2022-08-04T20:21:14.842213Z","iopub.execute_input":"2022-08-04T20:21:14.842749Z","iopub.status.idle":"2022-08-04T20:21:14.853830Z","shell.execute_reply.started":"2022-08-04T20:21:14.842714Z","shell.execute_reply":"2022-08-04T20:21:14.852585Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"There are so many missing values here we may want to just drop this and not worry about it, but this may also hold some signifance so I want do look at it before dropping","metadata":{}},{"cell_type":"code","source":"df.isna().any()","metadata":{"execution":{"iopub.status.busy":"2022-08-04T20:21:15.012171Z","iopub.execute_input":"2022-08-04T20:21:15.012807Z","iopub.status.idle":"2022-08-04T20:21:15.022204Z","shell.execute_reply.started":"2022-08-04T20:21:15.012767Z","shell.execute_reply":"2022-08-04T20:21:15.021157Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"This looks good for now, lets move onto looking at our categorical features","metadata":{}},{"cell_type":"markdown","source":"## Categorical","metadata":{}},{"cell_type":"code","source":"df.select_dtypes(include='object')","metadata":{"execution":{"iopub.status.busy":"2022-08-04T20:21:15.190955Z","iopub.execute_input":"2022-08-04T20:21:15.191353Z","iopub.status.idle":"2022-08-04T20:21:15.207334Z","shell.execute_reply.started":"2022-08-04T20:21:15.191321Z","shell.execute_reply":"2022-08-04T20:21:15.206603Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"These will need to be one hot encoded or dropped if we want to include them in our analysis","metadata":{}},{"cell_type":"markdown","source":"### Sex","metadata":{}},{"cell_type":"code","source":"df.Sex.describe()","metadata":{"execution":{"iopub.status.busy":"2022-08-04T20:21:15.360178Z","iopub.execute_input":"2022-08-04T20:21:15.360837Z","iopub.status.idle":"2022-08-04T20:21:15.388391Z","shell.execute_reply.started":"2022-08-04T20:21:15.360799Z","shell.execute_reply":"2022-08-04T20:21:15.387565Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.DataFrame(df.Sex.value_counts())","metadata":{"execution":{"iopub.status.busy":"2022-08-04T20:21:15.537845Z","iopub.execute_input":"2022-08-04T20:21:15.538591Z","iopub.status.idle":"2022-08-04T20:21:15.550961Z","shell.execute_reply.started":"2022-08-04T20:21:15.538538Z","shell.execute_reply":"2022-08-04T20:21:15.549911Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.DataFrame(df.Sex.value_counts(normalize=True))","metadata":{"execution":{"iopub.status.busy":"2022-08-04T20:21:15.716244Z","iopub.execute_input":"2022-08-04T20:21:15.716643Z","iopub.status.idle":"2022-08-04T20:21:15.728905Z","shell.execute_reply.started":"2022-08-04T20:21:15.716609Z","shell.execute_reply":"2022-08-04T20:21:15.727576Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.Sex.hist(density=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T20:21:15.887358Z","iopub.execute_input":"2022-08-04T20:21:15.887718Z","iopub.status.idle":"2022-08-04T20:21:16.010121Z","shell.execute_reply.started":"2022-08-04T20:21:15.887687Z","shell.execute_reply":"2022-08-04T20:21:16.009394Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Cabin\n\nThis will probably be dropped but I just want to look at it with some EDA","metadata":{}},{"cell_type":"code","source":"df.Cabin.value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-08-04T20:21:16.055328Z","iopub.execute_input":"2022-08-04T20:21:16.055849Z","iopub.status.idle":"2022-08-04T20:21:16.064775Z","shell.execute_reply.started":"2022-08-04T20:21:16.055814Z","shell.execute_reply":"2022-08-04T20:21:16.063871Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# get max index location of all the lengths in Cabin where its not NA\nlength = np.argmax([len(x) for x in df.Cabin[df.Cabin.isna() == False]])\n# retreive the data from that index location\ndf.Cabin[df.Cabin.isna() == False].iloc[length]","metadata":{"execution":{"iopub.status.busy":"2022-08-04T20:21:16.224180Z","iopub.execute_input":"2022-08-04T20:21:16.224562Z","iopub.status.idle":"2022-08-04T20:21:16.234050Z","shell.execute_reply.started":"2022-08-04T20:21:16.224528Z","shell.execute_reply":"2022-08-04T20:21:16.233289Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['Cabin_letter'] = df.Cabin.str.slice(0,1)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T20:21:16.387613Z","iopub.execute_input":"2022-08-04T20:21:16.388217Z","iopub.status.idle":"2022-08-04T20:21:16.397165Z","shell.execute_reply.started":"2022-08-04T20:21:16.388155Z","shell.execute_reply":"2022-08-04T20:21:16.395550Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.Cabin_letter.value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-08-04T20:21:16.561406Z","iopub.execute_input":"2022-08-04T20:21:16.561822Z","iopub.status.idle":"2022-08-04T20:21:16.570619Z","shell.execute_reply.started":"2022-08-04T20:21:16.561784Z","shell.execute_reply":"2022-08-04T20:21:16.569547Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Embarked","metadata":{}},{"cell_type":"code","source":"df.Embarked.value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-08-04T20:21:16.730054Z","iopub.execute_input":"2022-08-04T20:21:16.730592Z","iopub.status.idle":"2022-08-04T20:21:16.738027Z","shell.execute_reply.started":"2022-08-04T20:21:16.730558Z","shell.execute_reply":"2022-08-04T20:21:16.737356Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.Embarked.hist()","metadata":{"execution":{"iopub.status.busy":"2022-08-04T20:21:16.896070Z","iopub.execute_input":"2022-08-04T20:21:16.896612Z","iopub.status.idle":"2022-08-04T20:21:17.021322Z","shell.execute_reply.started":"2022-08-04T20:21:16.896576Z","shell.execute_reply":"2022-08-04T20:21:17.020528Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Looks like Sex and embarked could be one hot encoded, the other cateogrical data doesn't look too relavant. Although Cabin you may be able to extract the letter and see that feature has an impact on our model\n\nLets take a look at our numerics","metadata":{}},{"cell_type":"markdown","source":"## Numeric","metadata":{}},{"cell_type":"code","source":"cols = list(df.select_dtypes(include=['int', 'float']))\ncols","metadata":{"execution":{"iopub.status.busy":"2022-08-04T20:21:17.065232Z","iopub.execute_input":"2022-08-04T20:21:17.065750Z","iopub.status.idle":"2022-08-04T20:21:17.075142Z","shell.execute_reply.started":"2022-08-04T20:21:17.065716Z","shell.execute_reply":"2022-08-04T20:21:17.073522Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"helper.histos(df, cols)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T20:21:17.235088Z","iopub.execute_input":"2022-08-04T20:21:17.235713Z","iopub.status.idle":"2022-08-04T20:21:18.204179Z","shell.execute_reply.started":"2022-08-04T20:21:17.235675Z","shell.execute_reply":"2022-08-04T20:21:18.203043Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Looks like Pclass and Survived are categorical. \n\nPclass, Parch, and SibSp could be one hot encoded as they look like categorical data","metadata":{}},{"cell_type":"markdown","source":"### Survived","metadata":{}},{"cell_type":"code","source":"df.Survived.value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-08-04T20:21:18.206966Z","iopub.execute_input":"2022-08-04T20:21:18.207331Z","iopub.status.idle":"2022-08-04T20:21:18.215479Z","shell.execute_reply.started":"2022-08-04T20:21:18.207297Z","shell.execute_reply":"2022-08-04T20:21:18.214661Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.Survived.value_counts(normalize=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T20:21:18.217246Z","iopub.execute_input":"2022-08-04T20:21:18.217612Z","iopub.status.idle":"2022-08-04T20:21:18.229401Z","shell.execute_reply.started":"2022-08-04T20:21:18.217582Z","shell.execute_reply":"2022-08-04T20:21:18.228448Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# EDA Pairwise\n\nHere we will look at features and their relation to our target 'Survived'","metadata":{"tags":[]}},{"cell_type":"markdown","source":"## Data Transforms\n\nI would like to see an Age and Fare bracket to see any bigger trends so lets make them now","metadata":{}},{"cell_type":"code","source":"def age_bracket(age):\n    if age < 10:\n        return '0-10'\n    elif age >= 10 and age < 20:\n        return '10-20'\n    elif age >= 20 and age < 30:\n        return '20-30'\n    elif age >= 30 and age < 40:\n        return '30-40'\n    elif age >= 40 and age < 50:\n        return '40-50'\n    elif age >= 50 and age < 60:\n        return '50-60'\n    elif age >= 60 and age < 70:\n        return '60-70'\n    elif age >= 70:\n        return '70+'\n    \n    \ndf['Age_Bracket'] = df['Age_Gamma'].apply(age_bracket)\n# could also do this\n# df[\"Age_Bracket\"] = df[\"Age_Gamma\"] // 15 * 15","metadata":{"execution":{"iopub.status.busy":"2022-08-04T20:21:18.230590Z","iopub.execute_input":"2022-08-04T20:21:18.230938Z","iopub.status.idle":"2022-08-04T20:21:18.241617Z","shell.execute_reply.started":"2022-08-04T20:21:18.230897Z","shell.execute_reply":"2022-08-04T20:21:18.240849Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['Age_Bracket'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-08-04T20:21:18.243565Z","iopub.execute_input":"2022-08-04T20:21:18.244087Z","iopub.status.idle":"2022-08-04T20:21:18.254351Z","shell.execute_reply.started":"2022-08-04T20:21:18.244054Z","shell.execute_reply":"2022-08-04T20:21:18.253516Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def fare_bracket(fare):\n    if fare < 50:\n        return '0-50'\n    elif fare >= 50 and fare < 100:\n        return '50-100'\n    elif fare >= 100 and fare < 150:\n        return '100-150'\n    elif fare >= 150 and fare < 200:\n        return '150-200'\n    elif fare >= 200 and fare < 250:\n        return '200-250'\n    elif fare >= 250 and fare < 300:\n        return '250-300'\n    elif fare >= 300:\n        return '300+'\n    \n\ndf['Fare_Bracket'] = df['Fare'].apply(fare_bracket)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T20:21:18.255849Z","iopub.execute_input":"2022-08-04T20:21:18.256361Z","iopub.status.idle":"2022-08-04T20:21:18.266319Z","shell.execute_reply.started":"2022-08-04T20:21:18.256328Z","shell.execute_reply":"2022-08-04T20:21:18.265464Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['Fare_Bracket'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-08-04T20:21:18.268552Z","iopub.execute_input":"2022-08-04T20:21:18.268956Z","iopub.status.idle":"2022-08-04T20:21:18.278709Z","shell.execute_reply.started":"2022-08-04T20:21:18.268915Z","shell.execute_reply":"2022-08-04T20:21:18.277996Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Sex","metadata":{}},{"cell_type":"code","source":"pd.crosstab(df.Sex, df.Survived, margins=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T20:21:18.443781Z","iopub.execute_input":"2022-08-04T20:21:18.444325Z","iopub.status.idle":"2022-08-04T20:21:18.606766Z","shell.execute_reply.started":"2022-08-04T20:21:18.444283Z","shell.execute_reply":"2022-08-04T20:21:18.605726Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"There were almost twice as many males on board (577 to 314) but about twice as many females that survived (233 to 109)","metadata":{}},{"cell_type":"code","source":"pd.crosstab(df.Sex, df.Survived, margins=True, normalize='columns')","metadata":{"execution":{"iopub.status.busy":"2022-08-04T20:21:18.621215Z","iopub.execute_input":"2022-08-04T20:21:18.621545Z","iopub.status.idle":"2022-08-04T20:21:18.672907Z","shell.execute_reply.started":"2022-08-04T20:21:18.621516Z","shell.execute_reply":"2022-08-04T20:21:18.672201Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.crosstab(df.Sex, df.Survived, margins=True, normalize='index')","metadata":{"execution":{"iopub.status.busy":"2022-08-04T20:21:18.806856Z","iopub.execute_input":"2022-08-04T20:21:18.807378Z","iopub.status.idle":"2022-08-04T20:21:18.857842Z","shell.execute_reply.started":"2022-08-04T20:21:18.807343Z","shell.execute_reply":"2022-08-04T20:21:18.857132Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.crosstab(df.Sex, df.Survived, margins=True, normalize=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T20:21:18.992507Z","iopub.execute_input":"2022-08-04T20:21:18.993033Z","iopub.status.idle":"2022-08-04T20:21:19.046571Z","shell.execute_reply.started":"2022-08-04T20:21:18.993000Z","shell.execute_reply":"2022-08-04T20:21:19.045534Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def crossheatmap(data, cat1, cat2, margins=True, normalize=True, cmap='GnBu'):\n    table = pd.crosstab(data[cat1], data[cat2], margins=margins, normalize=True)\n    return sns.heatmap(table, annot=True, cmap=cmap)\n\ncrossheatmap(df, 'Sex', 'Survived', margins=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T20:21:19.163027Z","iopub.execute_input":"2022-08-04T20:21:19.163389Z","iopub.status.idle":"2022-08-04T20:21:19.381883Z","shell.execute_reply.started":"2022-08-04T20:21:19.163359Z","shell.execute_reply":"2022-08-04T20:21:19.380960Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Cabin","metadata":{}},{"cell_type":"code","source":"pd.crosstab(df.Cabin_letter, df.Survived, margins=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T20:21:19.383711Z","iopub.execute_input":"2022-08-04T20:21:19.384042Z","iopub.status.idle":"2022-08-04T20:21:19.433920Z","shell.execute_reply.started":"2022-08-04T20:21:19.384011Z","shell.execute_reply":"2022-08-04T20:21:19.433068Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.heatmap(pd.crosstab(df.Cabin_letter, df.Survived, margins=True, normalize='index'), annot=True, cmap='Blues')","metadata":{"execution":{"iopub.status.busy":"2022-08-04T20:21:19.514161Z","iopub.execute_input":"2022-08-04T20:21:19.514508Z","iopub.status.idle":"2022-08-04T20:21:19.777142Z","shell.execute_reply.started":"2022-08-04T20:21:19.514477Z","shell.execute_reply":"2022-08-04T20:21:19.776184Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.crosstab(df.Cabin_letter, df.Survived, margins=True, normalize=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T20:21:19.779073Z","iopub.execute_input":"2022-08-04T20:21:19.779412Z","iopub.status.idle":"2022-08-04T20:21:19.836652Z","shell.execute_reply.started":"2022-08-04T20:21:19.779380Z","shell.execute_reply":"2022-08-04T20:21:19.835860Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"crossheatmap(df, 'Cabin_letter', 'Survived')","metadata":{"execution":{"iopub.status.busy":"2022-08-04T20:21:19.865645Z","iopub.execute_input":"2022-08-04T20:21:19.866220Z","iopub.status.idle":"2022-08-04T20:21:20.141078Z","shell.execute_reply.started":"2022-08-04T20:21:19.866184Z","shell.execute_reply":"2022-08-04T20:21:20.140282Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Don't see too much here except that if your cabin was later in the alphabet then you were less likely to survive, however the A cabin doesn't show that trend in its class","metadata":{}},{"cell_type":"markdown","source":"## Embarked","metadata":{}},{"cell_type":"code","source":"pd.crosstab(df.Embarked, df.Survived, margins=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T20:21:20.142397Z","iopub.execute_input":"2022-08-04T20:21:20.142817Z","iopub.status.idle":"2022-08-04T20:21:20.190314Z","shell.execute_reply.started":"2022-08-04T20:21:20.142785Z","shell.execute_reply":"2022-08-04T20:21:20.189347Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.crosstab(df.Embarked, df.Survived, margins=True, normalize=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T20:21:20.226421Z","iopub.execute_input":"2022-08-04T20:21:20.226822Z","iopub.status.idle":"2022-08-04T20:21:20.281996Z","shell.execute_reply.started":"2022-08-04T20:21:20.226786Z","shell.execute_reply":"2022-08-04T20:21:20.280858Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.heatmap(pd.crosstab(df.Embarked, df.Survived, margins=True, normalize=True), annot=True, cmap='Blues')","metadata":{"execution":{"iopub.status.busy":"2022-08-04T20:21:20.396229Z","iopub.execute_input":"2022-08-04T20:21:20.396620Z","iopub.status.idle":"2022-08-04T20:21:20.621197Z","shell.execute_reply.started":"2022-08-04T20:21:20.396583Z","shell.execute_reply":"2022-08-04T20:21:20.620201Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"people in S were about twice as likely to not survive (.48 to .24) but this is hard to tell because its normalized for all people","metadata":{}},{"cell_type":"code","source":"pd.crosstab(df.Embarked, df.Survived, margins=True, normalize='index')","metadata":{"execution":{"iopub.status.busy":"2022-08-04T20:21:20.623119Z","iopub.execute_input":"2022-08-04T20:21:20.623424Z","iopub.status.idle":"2022-08-04T20:21:20.674936Z","shell.execute_reply.started":"2022-08-04T20:21:20.623394Z","shell.execute_reply":"2022-08-04T20:21:20.673920Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"This tells a similar story but also that people in C had the highest chance of survival as 0.55","metadata":{}},{"cell_type":"code","source":"ax = sns.barplot(x='Embarked', y='Survived', data=df)\n# for i in ax.containers:\n#     ax.bar_label(i,)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T20:21:20.743637Z","iopub.execute_input":"2022-08-04T20:21:20.743995Z","iopub.status.idle":"2022-08-04T20:21:20.947451Z","shell.execute_reply.started":"2022-08-04T20:21:20.743965Z","shell.execute_reply":"2022-08-04T20:21:20.946257Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## SibSp","metadata":{}},{"cell_type":"code","source":"pd.crosstab(df.SibSp, df.Survived, margins=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T20:21:20.949571Z","iopub.execute_input":"2022-08-04T20:21:20.949988Z","iopub.status.idle":"2022-08-04T20:21:21.001362Z","shell.execute_reply.started":"2022-08-04T20:21:20.949936Z","shell.execute_reply":"2022-08-04T20:21:21.000289Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.heatmap(pd.crosstab(df.SibSp, df.Survived, margins=True, normalize='index'), annot=True, cmap='Blues')","metadata":{"execution":{"iopub.status.busy":"2022-08-04T20:21:21.093049Z","iopub.execute_input":"2022-08-04T20:21:21.093492Z","iopub.status.idle":"2022-08-04T20:21:21.338859Z","shell.execute_reply.started":"2022-08-04T20:21:21.093455Z","shell.execute_reply":"2022-08-04T20:21:21.337920Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"This seems to show that the small numbers 0,1,2 had the highest chance of survival","metadata":{}},{"cell_type":"code","source":"sns.heatmap(pd.crosstab(df.SibSp, df.Survived, margins=True, normalize=True), annot=True, cmap='Blues')","metadata":{"execution":{"iopub.status.busy":"2022-08-04T20:21:21.340470Z","iopub.execute_input":"2022-08-04T20:21:21.340771Z","iopub.status.idle":"2022-08-04T20:21:21.621563Z","shell.execute_reply.started":"2022-08-04T20:21:21.340741Z","shell.execute_reply":"2022-08-04T20:21:21.620747Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.groupby('SibSp')['Survived'].mean().plot(grid=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T20:21:21.623532Z","iopub.execute_input":"2022-08-04T20:21:21.624150Z","iopub.status.idle":"2022-08-04T20:21:21.797138Z","shell.execute_reply.started":"2022-08-04T20:21:21.624087Z","shell.execute_reply":"2022-08-04T20:21:21.796248Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Parch","metadata":{}},{"cell_type":"code","source":"pd.crosstab(df.Parch, df.Survived, margins=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T20:21:21.798948Z","iopub.execute_input":"2022-08-04T20:21:21.799596Z","iopub.status.idle":"2022-08-04T20:21:21.848148Z","shell.execute_reply.started":"2022-08-04T20:21:21.799540Z","shell.execute_reply":"2022-08-04T20:21:21.847073Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.heatmap(pd.crosstab(df.Parch, df.Survived, margins=True, normalize='index'), annot=True, cmap='Blues')","metadata":{"execution":{"iopub.status.busy":"2022-08-04T20:21:21.849557Z","iopub.execute_input":"2022-08-04T20:21:21.850160Z","iopub.status.idle":"2022-08-04T20:21:22.098062Z","shell.execute_reply.started":"2022-08-04T20:21:21.850096Z","shell.execute_reply":"2022-08-04T20:21:22.097274Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Seems like the more parents/children the more likely you will not survived","metadata":{}},{"cell_type":"code","source":"df.groupby('Parch')['Survived'].mean().plot(grid=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T20:21:22.099565Z","iopub.execute_input":"2022-08-04T20:21:22.100023Z","iopub.status.idle":"2022-08-04T20:21:22.262187Z","shell.execute_reply.started":"2022-08-04T20:21:22.099987Z","shell.execute_reply":"2022-08-04T20:21:22.261222Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Pclass","metadata":{}},{"cell_type":"code","source":"df.groupby('Pclass')['Survived'].mean()","metadata":{"execution":{"iopub.status.busy":"2022-08-04T20:21:22.264276Z","iopub.execute_input":"2022-08-04T20:21:22.264587Z","iopub.status.idle":"2022-08-04T20:21:22.275188Z","shell.execute_reply.started":"2022-08-04T20:21:22.264555Z","shell.execute_reply":"2022-08-04T20:21:22.274032Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.groupby('Pclass')['Survived'].mean().plot(grid=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T20:21:22.299195Z","iopub.execute_input":"2022-08-04T20:21:22.299560Z","iopub.status.idle":"2022-08-04T20:21:22.473077Z","shell.execute_reply.started":"2022-08-04T20:21:22.299529Z","shell.execute_reply":"2022-08-04T20:21:22.472204Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Age","metadata":{}},{"cell_type":"code","source":"df.describe()","metadata":{"execution":{"iopub.status.busy":"2022-08-04T20:21:22.478092Z","iopub.execute_input":"2022-08-04T20:21:22.478420Z","iopub.status.idle":"2022-08-04T20:21:22.508940Z","shell.execute_reply.started":"2022-08-04T20:21:22.478391Z","shell.execute_reply":"2022-08-04T20:21:22.507938Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"helper.multiboxplot(df, 'Survived', 'Age_Gamma')","metadata":{"execution":{"iopub.status.busy":"2022-08-04T20:21:22.645915Z","iopub.execute_input":"2022-08-04T20:21:22.646282Z","iopub.status.idle":"2022-08-04T20:21:22.798080Z","shell.execute_reply.started":"2022-08-04T20:21:22.646251Z","shell.execute_reply":"2022-08-04T20:21:22.797158Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"helper.lowess_scatter(df, 'Age_Gamma', 'Survived')","metadata":{"execution":{"iopub.status.busy":"2022-08-04T20:21:22.813065Z","iopub.execute_input":"2022-08-04T20:21:22.813475Z","iopub.status.idle":"2022-08-04T20:21:23.009001Z","shell.execute_reply.started":"2022-08-04T20:21:22.813437Z","shell.execute_reply":"2022-08-04T20:21:23.007882Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['Age_Bracket'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-08-04T20:21:23.011292Z","iopub.execute_input":"2022-08-04T20:21:23.011627Z","iopub.status.idle":"2022-08-04T20:21:23.020234Z","shell.execute_reply.started":"2022-08-04T20:21:23.011593Z","shell.execute_reply":"2022-08-04T20:21:23.019193Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df[[\"Age_Bracket\", \"Survived\"]].groupby(['Age_Bracket']).mean()","metadata":{"execution":{"iopub.status.busy":"2022-08-04T20:21:23.165423Z","iopub.execute_input":"2022-08-04T20:21:23.166126Z","iopub.status.idle":"2022-08-04T20:21:23.179255Z","shell.execute_reply.started":"2022-08-04T20:21:23.166069Z","shell.execute_reply":"2022-08-04T20:21:23.178025Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.groupby('Age_Bracket')['Survived'].mean().plot()","metadata":{"execution":{"iopub.status.busy":"2022-08-04T20:21:23.338240Z","iopub.execute_input":"2022-08-04T20:21:23.338756Z","iopub.status.idle":"2022-08-04T20:21:23.605069Z","shell.execute_reply.started":"2022-08-04T20:21:23.338720Z","shell.execute_reply":"2022-08-04T20:21:23.604236Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"This looks like a pretty strong predictor","metadata":{}},{"cell_type":"code","source":"sns.heatmap(pd.crosstab(df.Age_Bracket, df.Survived, margins=True, normalize='index'), annot=True, cmap='Blues')","metadata":{"execution":{"iopub.status.busy":"2022-08-04T20:21:23.606381Z","iopub.execute_input":"2022-08-04T20:21:23.606793Z","iopub.status.idle":"2022-08-04T20:21:23.869449Z","shell.execute_reply.started":"2022-08-04T20:21:23.606762Z","shell.execute_reply":"2022-08-04T20:21:23.868685Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"stats.pearsonr(df.Age_Gamma, df.Survived)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T20:21:23.871327Z","iopub.execute_input":"2022-08-04T20:21:23.871914Z","iopub.status.idle":"2022-08-04T20:21:23.880757Z","shell.execute_reply.started":"2022-08-04T20:21:23.871870Z","shell.execute_reply":"2022-08-04T20:21:23.879691Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Fare","metadata":{}},{"cell_type":"code","source":"df.groupby('Fare_Bracket')['Survived'].mean()","metadata":{"execution":{"iopub.status.busy":"2022-08-04T20:21:23.882444Z","iopub.execute_input":"2022-08-04T20:21:23.882740Z","iopub.status.idle":"2022-08-04T20:21:23.893006Z","shell.execute_reply.started":"2022-08-04T20:21:23.882712Z","shell.execute_reply":"2022-08-04T20:21:23.892210Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df[[\"Fare_Bracket\", \"Survived\"]].groupby(['Fare_Bracket']).mean()","metadata":{"execution":{"iopub.status.busy":"2022-08-04T20:21:24.026196Z","iopub.execute_input":"2022-08-04T20:21:24.026838Z","iopub.status.idle":"2022-08-04T20:21:24.041506Z","shell.execute_reply.started":"2022-08-04T20:21:24.026775Z","shell.execute_reply":"2022-08-04T20:21:24.040291Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.groupby('Fare_Bracket')['Survived'].mean().plot()","metadata":{"execution":{"iopub.status.busy":"2022-08-04T20:21:24.198300Z","iopub.execute_input":"2022-08-04T20:21:24.198713Z","iopub.status.idle":"2022-08-04T20:21:24.339329Z","shell.execute_reply.started":"2022-08-04T20:21:24.198675Z","shell.execute_reply":"2022-08-04T20:21:24.338411Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.heatmap(df.corr(), annot=True, cmap='Blues')","metadata":{"execution":{"iopub.status.busy":"2022-08-04T20:21:24.369506Z","iopub.execute_input":"2022-08-04T20:21:24.369865Z","iopub.status.idle":"2022-08-04T20:21:24.656378Z","shell.execute_reply.started":"2022-08-04T20:21:24.369834Z","shell.execute_reply":"2022-08-04T20:21:24.655519Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"stats.pearsonr(df.Fare, df.Survived)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T20:21:24.658363Z","iopub.execute_input":"2022-08-04T20:21:24.658643Z","iopub.status.idle":"2022-08-04T20:21:24.665283Z","shell.execute_reply.started":"2022-08-04T20:21:24.658615Z","shell.execute_reply":"2022-08-04T20:21:24.664334Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Prepare data for ML Algorithms","metadata":{}},{"cell_type":"markdown","source":"## Data Cleansing II","metadata":{}},{"cell_type":"code","source":"df","metadata":{"execution":{"iopub.status.busy":"2022-08-04T20:21:24.713017Z","iopub.execute_input":"2022-08-04T20:21:24.713381Z","iopub.status.idle":"2022-08-04T20:21:24.739502Z","shell.execute_reply.started":"2022-08-04T20:21:24.713350Z","shell.execute_reply":"2022-08-04T20:21:24.738513Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# drop unused features\ndf.drop(['Cabin', 'Cabin_letter', 'Age_Bracket', 'Fare_Bracket'], axis=1, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T20:21:24.882773Z","iopub.execute_input":"2022-08-04T20:21:24.883151Z","iopub.status.idle":"2022-08-04T20:21:24.889373Z","shell.execute_reply.started":"2022-08-04T20:21:24.883119Z","shell.execute_reply":"2022-08-04T20:21:24.888146Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df","metadata":{"execution":{"iopub.status.busy":"2022-08-04T20:21:25.060935Z","iopub.execute_input":"2022-08-04T20:21:25.061329Z","iopub.status.idle":"2022-08-04T20:21:25.084468Z","shell.execute_reply.started":"2022-08-04T20:21:25.061295Z","shell.execute_reply":"2022-08-04T20:21:25.083207Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## One Hot Encoding","metadata":{}},{"cell_type":"code","source":"from sklearn.preprocessing import OneHotEncoder\n\ncat_attribs = ['Pclass', 'Sex', 'SibSp', 'Parch', 'Embarked']\n\nX_train_cat = df[cat_attribs].copy()\n\nohe = OneHotEncoder()\n\nohe_x_train = ohe.fit_transform(X_train_cat)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T20:21:25.232430Z","iopub.execute_input":"2022-08-04T20:21:25.232795Z","iopub.status.idle":"2022-08-04T20:21:25.270191Z","shell.execute_reply.started":"2022-08-04T20:21:25.232762Z","shell.execute_reply":"2022-08-04T20:21:25.269164Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ohe.categories_","metadata":{"execution":{"iopub.status.busy":"2022-08-04T20:21:25.399741Z","iopub.execute_input":"2022-08-04T20:21:25.400134Z","iopub.status.idle":"2022-08-04T20:21:25.407548Z","shell.execute_reply.started":"2022-08-04T20:21:25.400085Z","shell.execute_reply":"2022-08-04T20:21:25.406227Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ohe_x_train.toarray()","metadata":{"execution":{"iopub.status.busy":"2022-08-04T20:21:25.568991Z","iopub.execute_input":"2022-08-04T20:21:25.569558Z","iopub.status.idle":"2022-08-04T20:21:25.575665Z","shell.execute_reply.started":"2022-08-04T20:21:25.569511Z","shell.execute_reply":"2022-08-04T20:21:25.574879Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# cols = ohe.get_feature_names_out()\n# pd.DataFrame(ohe_x_train.toarray(), columns=cols)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T20:21:25.736063Z","iopub.execute_input":"2022-08-04T20:21:25.736595Z","iopub.status.idle":"2022-08-04T20:21:25.740125Z","shell.execute_reply.started":"2022-08-04T20:21:25.736561Z","shell.execute_reply":"2022-08-04T20:21:25.739234Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Split Dataset","metadata":{}},{"cell_type":"code","source":"features_to_drop = ['Name', 'Ticket', 'PassengerId', 'Cabin']\nall_features = list(orig)\nfeatures = [i for i in all_features if i not in features_to_drop]","metadata":{"execution":{"iopub.status.busy":"2022-08-04T20:21:25.904746Z","iopub.execute_input":"2022-08-04T20:21:25.905273Z","iopub.status.idle":"2022-08-04T20:21:25.910420Z","shell.execute_reply.started":"2022-08-04T20:21:25.905238Z","shell.execute_reply":"2022-08-04T20:21:25.909288Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"features.remove('Survived')\nlabels = ['Survived']\n\nprint(features)\nprint(labels)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T20:21:26.089154Z","iopub.execute_input":"2022-08-04T20:21:26.089666Z","iopub.status.idle":"2022-08-04T20:21:26.095213Z","shell.execute_reply.started":"2022-08-04T20:21:26.089633Z","shell.execute_reply":"2022-08-04T20:21:26.094357Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"orig[labels]","metadata":{"execution":{"iopub.status.busy":"2022-08-04T20:21:26.271654Z","iopub.execute_input":"2022-08-04T20:21:26.272196Z","iopub.status.idle":"2022-08-04T20:21:26.283596Z","shell.execute_reply.started":"2022-08-04T20:21:26.272159Z","shell.execute_reply":"2022-08-04T20:21:26.282775Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data","metadata":{"execution":{"iopub.status.busy":"2022-08-04T20:21:26.405303Z","iopub.execute_input":"2022-08-04T20:21:26.405657Z","iopub.status.idle":"2022-08-04T20:21:26.428828Z","shell.execute_reply.started":"2022-08-04T20:21:26.405627Z","shell.execute_reply":"2022-08-04T20:21:26.427864Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\n# pass in pandas dataframe\n# X = orig[features]\n# y = orig[labels]\n\n# input pandas dataframes\n# X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.15)\n\n# kaggle dataset only\nX_train = orig[features]\ny_train = orig[labels]\n\nX_test = test_data[features]","metadata":{"execution":{"iopub.status.busy":"2022-08-04T20:21:26.566780Z","iopub.execute_input":"2022-08-04T20:21:26.567189Z","iopub.status.idle":"2022-08-04T20:21:26.626612Z","shell.execute_reply.started":"2022-08-04T20:21:26.567152Z","shell.execute_reply":"2022-08-04T20:21:26.625483Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Create Pipeline","metadata":{}},{"cell_type":"code","source":"from sklearn.pipeline import Pipeline\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.compose import ColumnTransformer\n\nfrom sklearn.base import BaseEstimator, TransformerMixin","metadata":{"execution":{"iopub.status.busy":"2022-08-04T20:21:26.757305Z","iopub.execute_input":"2022-08-04T20:21:26.757861Z","iopub.status.idle":"2022-08-04T20:21:26.933389Z","shell.execute_reply.started":"2022-08-04T20:21:26.757809Z","shell.execute_reply":"2022-08-04T20:21:26.932253Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class columnDropperTransformer():\n    def __init__(self,columns):\n        self.columns=columns\n\n    def fit(self, X, y=None):\n        return self \n    \n    def transform(self,X,y=None):\n        return X.drop(self.columns,axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T20:21:26.953310Z","iopub.execute_input":"2022-08-04T20:21:26.953898Z","iopub.status.idle":"2022-08-04T20:21:26.960623Z","shell.execute_reply.started":"2022-08-04T20:21:26.953843Z","shell.execute_reply":"2022-08-04T20:21:26.959641Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class GammaTransformer(BaseEstimator, TransformerMixin):\n    \n    def __init__(self, col):\n        self.col = col\n        \n    def fit(self, X, y=None):\n        return self\n    \n    def transform(self, X, y=None):\n        X = X.copy()\n        X[self.col] = X[self.col].fillna(add_gammas(X, self.col), inplace=False)\n        return X","metadata":{"execution":{"iopub.status.busy":"2022-08-04T20:21:27.173877Z","iopub.execute_input":"2022-08-04T20:21:27.174282Z","iopub.status.idle":"2022-08-04T20:21:27.180653Z","shell.execute_reply.started":"2022-08-04T20:21:27.174244Z","shell.execute_reply":"2022-08-04T20:21:27.179679Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# unit test gamma function\ntest = orig.copy()\ntest['AgeNew'] = test['Age'].fillna(add_gammas(test,'Age'), inplace=False)\ntest[test.Age.isnull()].head(5)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T20:21:27.379791Z","iopub.execute_input":"2022-08-04T20:21:27.380472Z","iopub.status.idle":"2022-08-04T20:21:27.408616Z","shell.execute_reply.started":"2022-08-04T20:21:27.380431Z","shell.execute_reply":"2022-08-04T20:21:27.407854Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train.dtypes","metadata":{"execution":{"iopub.status.busy":"2022-08-04T20:21:27.575161Z","iopub.execute_input":"2022-08-04T20:21:27.575695Z","iopub.status.idle":"2022-08-04T20:21:27.583416Z","shell.execute_reply.started":"2022-08-04T20:21:27.575660Z","shell.execute_reply":"2022-08-04T20:21:27.582600Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train.isna().any()","metadata":{"execution":{"iopub.status.busy":"2022-08-04T20:21:27.815630Z","iopub.execute_input":"2022-08-04T20:21:27.815995Z","iopub.status.idle":"2022-08-04T20:21:27.826645Z","shell.execute_reply.started":"2022-08-04T20:21:27.815964Z","shell.execute_reply":"2022-08-04T20:21:27.825451Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"num_attribs = ['Fare', 'Age','SibSp', 'Parch']\ncat_attribs = ['Sex','Pclass', 'Embarked']    # +2 +3 +7 +7 +3\n\n# impute embarked and ohe\ncat_pipe = Pipeline([\n    ('imputer', SimpleImputer(strategy='most_frequent')), \n    (\"ohe\", OneHotEncoder(sparse=False))\n])\n\n# impute age w mean and scale\nnum_pipe = Pipeline([\n    ('imputer', SimpleImputer(strategy='mean')),           \n    ('scaler', StandardScaler())\n])\n\n# impute age with gamma and scale\nnum_pipe2 = Pipeline([\n    ('imputerAge', GammaTransformer(col='Age')),\n    ('imputer', SimpleImputer(strategy='mean')),\n    ('scaler', StandardScaler())\n])\n\n# not used\ndrop_pipe = Pipeline([\n    ('dropper', columnDropperTransformer(features_to_drop))\n])\n\n# create full preprocessor pipeline\npreprocessor = ColumnTransformer([\n        (\"num\", num_pipe2, num_attribs),       \n        (\"cat\", cat_pipe, cat_attribs),   \n    ])\nprint('done')","metadata":{"execution":{"iopub.status.busy":"2022-08-04T20:28:06.302616Z","iopub.execute_input":"2022-08-04T20:28:06.303310Z","iopub.status.idle":"2022-08-04T20:28:06.313742Z","shell.execute_reply.started":"2022-08-04T20:28:06.303271Z","shell.execute_reply":"2022-08-04T20:28:06.312561Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Fit Transform","metadata":{}},{"cell_type":"code","source":"titanic_prepared = preprocessor.fit(X_train)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T20:28:08.765488Z","iopub.execute_input":"2022-08-04T20:28:08.765864Z","iopub.status.idle":"2022-08-04T20:28:08.794052Z","shell.execute_reply.started":"2022-08-04T20:28:08.765828Z","shell.execute_reply":"2022-08-04T20:28:08.793188Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# cols = titanic_prepared.transformers[1][1].named_steps['ohe'].get_feature_names()\n# cols = list(cols)\n# columns = num_attribs + list(cols)\n# print(columns)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T20:28:09.565674Z","iopub.execute_input":"2022-08-04T20:28:09.566039Z","iopub.status.idle":"2022-08-04T20:28:09.588104Z","shell.execute_reply.started":"2022-08-04T20:28:09.566008Z","shell.execute_reply":"2022-08-04T20:28:09.586870Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"titanic_prepared = preprocessor.transform(X_train)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T20:28:22.458362Z","iopub.execute_input":"2022-08-04T20:28:22.458922Z","iopub.status.idle":"2022-08-04T20:28:22.475790Z","shell.execute_reply.started":"2022-08-04T20:28:22.458888Z","shell.execute_reply":"2022-08-04T20:28:22.474863Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.set_option('display.max_columns', None)\npd.DataFrame(titanic_prepared).head(10)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T20:28:23.858777Z","iopub.execute_input":"2022-08-04T20:28:23.859486Z","iopub.status.idle":"2022-08-04T20:28:23.881050Z","shell.execute_reply.started":"2022-08-04T20:28:23.859445Z","shell.execute_reply":"2022-08-04T20:28:23.880279Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train.head(10)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T20:28:28.166152Z","iopub.execute_input":"2022-08-04T20:28:28.166515Z","iopub.status.idle":"2022-08-04T20:28:28.182308Z","shell.execute_reply.started":"2022-08-04T20:28:28.166483Z","shell.execute_reply":"2022-08-04T20:28:28.181130Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Modeling","metadata":{}},{"cell_type":"code","source":"from sklearn.metrics import precision_score\nfrom sklearn.metrics import recall_score\nfrom sklearn.metrics import f1_score\nfrom sklearn.metrics import accuracy_score\n\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.neighbors import KNeighborsClassifier\nfrom sklearn.svm import SVC\nfrom sklearn.tree import DecisionTreeClassifier\nfrom sklearn.ensemble import RandomForestClassifier, AdaBoostClassifier, GradientBoostingClassifier\nfrom sklearn.neural_network import MLPClassifier\nfrom sklearn.naive_bayes import GaussianNB\nfrom xgboost import XGBClassifier\nimport warnings\n\nwarnings.filterwarnings(\"ignore\", category=UserWarning)\n\nnames = ['Logistic Reg',\n    'Nearest Neighbors',\n    'SVC',\n    'Decision Tree',\n    'Neural Net', \n    'AdaBoost',\n    'Naive Bayes',\n    'Random Forest', \n    'gboost']\n\nclassifiers = [\n    LogisticRegression(),\n    KNeighborsClassifier(3),\n    SVC(gamma='auto'),\n    DecisionTreeClassifier(max_depth=10),\n    MLPClassifier(alpha=1, max_iter=1000),\n    AdaBoostClassifier(),\n    GaussianNB(),\n    RandomForestClassifier(max_depth=10, n_estimators=100, max_features=3),\n    GradientBoostingClassifier()\n]","metadata":{"execution":{"iopub.status.busy":"2022-08-04T20:28:29.241706Z","iopub.execute_input":"2022-08-04T20:28:29.242285Z","iopub.status.idle":"2022-08-04T20:28:29.254865Z","shell.execute_reply.started":"2022-08-04T20:28:29.242231Z","shell.execute_reply":"2022-08-04T20:28:29.253742Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"res_train = {}\nfor name, clf in zip(names, classifiers):\n    clf.fit(titanic_prepared, y_train.values.ravel())\n    preds = clf.predict(titanic_prepared)\n    scores = []\n    scores.append(accuracy_score(y_train, preds))\n    scores.append(precision_score(y_train, preds))\n    scores.append(recall_score(y_train, preds))\n    scores.append(f1_score(y_train, preds))\n    res_train[name] = scores\n    \n# res_sorted = dict(sorted(res_train.items(), key=lambda x: x[1], reverse=True))\ndf_scores = pd.DataFrame.from_dict(res_train)\ndf_scores = df_scores.T\ndf_scores.columns = ['accuracy', 'precision', 'recall', 'f1']\ndf_scores.reset_index()\ndf_scores.style.background_gradient(cmap='Greens', axis=0)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T20:28:30.273300Z","iopub.execute_input":"2022-08-04T20:28:30.273687Z","iopub.status.idle":"2022-08-04T20:28:31.900893Z","shell.execute_reply.started":"2022-08-04T20:28:30.273641Z","shell.execute_reply":"2022-08-04T20:28:31.900202Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Decision tree, random forest and gboost all look good lets use those in cv","metadata":{}},{"cell_type":"markdown","source":"## Cross Validation","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import StratifiedKFold\nfrom sklearn.model_selection import cross_validate\nfrom sklearn.model_selection import cross_val_score\nfrom sklearn.model_selection import cross_val_predict","metadata":{"execution":{"iopub.status.busy":"2022-08-04T20:28:41.919814Z","iopub.execute_input":"2022-08-04T20:28:41.920349Z","iopub.status.idle":"2022-08-04T20:28:41.925725Z","shell.execute_reply.started":"2022-08-04T20:28:41.920309Z","shell.execute_reply":"2022-08-04T20:28:41.924504Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rf_clf = RandomForestClassifier(max_depth=10, n_estimators=100, max_features=3, random_state=42)\nrf_clf.fit(titanic_prepared, y_train.values.ravel())\n\ngb_clf = GradientBoostingClassifier(random_state=42)\ngb_clf.fit(titanic_prepared, y_train.values.ravel())\n\ndt_clf = DecisionTreeClassifier(max_depth=10, random_state=42)\ndt_clf.fit(titanic_prepared, y_train.values.ravel())\nprint('done')","metadata":{"execution":{"iopub.status.busy":"2022-08-04T20:28:44.784850Z","iopub.execute_input":"2022-08-04T20:28:44.786375Z","iopub.status.idle":"2022-08-04T20:28:45.129167Z","shell.execute_reply.started":"2022-08-04T20:28:44.786321Z","shell.execute_reply":"2022-08-04T20:28:45.128079Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"metrics = ['accuracy', 'precision', 'recall', 'f1', 'roc_auc']\nfolds = StratifiedKFold(n_splits=10, shuffle=True)\n\nscores_rf = cross_validate(rf_clf, titanic_prepared, y_train.values.ravel(), scoring=metrics, cv=folds, return_train_score=True, n_jobs=-1)\nscores_gb = cross_validate(gb_clf, titanic_prepared, y_train.values.ravel(), scoring=metrics, cv=folds, return_train_score=True, n_jobs=-1)\nscores_dt = cross_validate(dt_clf, titanic_prepared, y_train.values.ravel(), scoring=metrics, cv=folds, return_train_score=True, n_jobs=-1)\nprint('done')","metadata":{"execution":{"iopub.status.busy":"2022-08-04T20:28:47.029792Z","iopub.execute_input":"2022-08-04T20:28:47.030515Z","iopub.status.idle":"2022-08-04T20:28:48.867647Z","shell.execute_reply.started":"2022-08-04T20:28:47.030462Z","shell.execute_reply":"2022-08-04T20:28:48.866675Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, (ax1, ax2) = plt.subplots(1,2, figsize=(15,5))\n\nmetric = 'accuracy'\n\nax1.plot(scores_rf['train_'+metric],'g--', label='random forest',)\nax1.plot(scores_gb['train_'+metric],'b--', label='gradient boost')\nax1.plot(scores_dt['train_'+metric],'r--', label='decision tree')\nax1.set_title('Train')\n\nax2.plot(scores_rf['test_'+metric], label='random forest', c='g')\nax2.plot(scores_gb['test_'+metric], label='gradient boost', c='b')\nax2.plot(scores_dt['test_'+metric], label='decision tree', c='r')\n# ax2.sharey(ax1)\nax2.set_title('Validation')\n\nfig.legend()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-04T20:28:50.734742Z","iopub.execute_input":"2022-08-04T20:28:50.735126Z","iopub.status.idle":"2022-08-04T20:28:51.059564Z","shell.execute_reply.started":"2022-08-04T20:28:50.735062Z","shell.execute_reply":"2022-08-04T20:28:51.058566Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(10, 6))\nscore1 = scores_rf['test_'+metric]\nscore2 = scores_gb['test_'+metric]\nscore3 = scores_dt['test_'+metric]\nplt.plot([1]*10, score1, \".\")\nplt.plot([2]*10, score2, \".\")\nplt.plot([3]*10, score3, \".\")\nplt.boxplot([score1, score2, score3], labels=(\"random forest\",\"gradient boost\", 'decision tree'))\nplt.ylabel(\"Accuracy\", fontsize=14)\nplt.title('cross val test {} scores'.format(metric))\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-04T20:28:55.191289Z","iopub.execute_input":"2022-08-04T20:28:55.191806Z","iopub.status.idle":"2022-08-04T20:28:55.362235Z","shell.execute_reply.started":"2022-08-04T20:28:55.191773Z","shell.execute_reply":"2022-08-04T20:28:55.360942Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Random forest has smaller error bars and the highest value so lets tune that ","metadata":{}},{"cell_type":"markdown","source":"## Hyperparameter Tuning","metadata":{}},{"cell_type":"markdown","source":"### Random Forest","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import GridSearchCV\n\nparams = [{\n    'n_estimators':[50,100,200,300],\n    'max_features':[1,2,3],\n    'max_depth':[2,5,8,10]\n}]\n\n\ngrd_search_rf = GridSearchCV(rf_clf,\n                         param_grid=params,\n                         cv=folds,\n                         scoring='f1',\n                         n_jobs=-1)\n\n\ngrd_search_rf.fit(titanic_prepared, y_train.values.ravel())","metadata":{"execution":{"iopub.status.busy":"2022-08-04T20:29:05.563134Z","iopub.execute_input":"2022-08-04T20:29:05.563829Z","iopub.status.idle":"2022-08-04T20:30:05.199725Z","shell.execute_reply.started":"2022-08-04T20:29:05.563782Z","shell.execute_reply":"2022-08-04T20:30:05.198603Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.DataFrame(grd_search_rf.cv_results_).sort_values(by='rank_test_score').head(5)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T20:30:05.202153Z","iopub.execute_input":"2022-08-04T20:30:05.202477Z","iopub.status.idle":"2022-08-04T20:30:05.234852Z","shell.execute_reply.started":"2022-08-04T20:30:05.202443Z","shell.execute_reply":"2022-08-04T20:30:05.233901Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Gradient Boosting Classifier","metadata":{}},{"cell_type":"code","source":"params = [{\n    'n_estimators':[50,100,200,300],\n    'learning_rate':[.01, .1, 1],\n    'max_depth':[2,5,8,10]\n}]\n\n\ngrd_search_gb = GridSearchCV(gb_clf,\n                         param_grid=params,\n                         cv=folds,\n                         scoring='f1',\n                         n_jobs=-1)\n\n\ngrd_search_gb.fit(titanic_prepared, y_train.values.ravel())","metadata":{"execution":{"iopub.status.busy":"2022-08-04T20:30:05.236519Z","iopub.execute_input":"2022-08-04T20:30:05.236832Z","iopub.status.idle":"2022-08-04T20:32:04.294379Z","shell.execute_reply.started":"2022-08-04T20:30:05.236800Z","shell.execute_reply":"2022-08-04T20:32:04.293360Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.DataFrame(grd_search_gb.cv_results_).sort_values(by='rank_test_score').head(5)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T20:32:04.295962Z","iopub.execute_input":"2022-08-04T20:32:04.296263Z","iopub.status.idle":"2022-08-04T20:32:04.328294Z","shell.execute_reply.started":"2022-08-04T20:32:04.296233Z","shell.execute_reply":"2022-08-04T20:32:04.327209Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(grd_search_rf.best_estimator_)\nprint(grd_search_rf.best_params_)\nprint(grd_search_rf.best_score_)\nprint()\nprint(grd_search_gb.best_estimator_)\nprint(grd_search_gb.best_params_)\nprint(grd_search_gb.best_score_)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T20:32:04.330583Z","iopub.execute_input":"2022-08-04T20:32:04.331249Z","iopub.status.idle":"2022-08-04T20:32:04.340902Z","shell.execute_reply.started":"2022-08-04T20:32:04.331176Z","shell.execute_reply":"2022-08-04T20:32:04.339471Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Testing","metadata":{}},{"cell_type":"code","source":"X_test.isna().any()","metadata":{"execution":{"iopub.status.busy":"2022-08-04T20:32:15.971597Z","iopub.execute_input":"2022-08-04T20:32:15.971979Z","iopub.status.idle":"2022-08-04T20:32:15.980320Z","shell.execute_reply.started":"2022-08-04T20:32:15.971946Z","shell.execute_reply":"2022-08-04T20:32:15.979164Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import confusion_matrix\nfrom sklearn.metrics import classification_report\n\n# transform test data do NOT fit\nX_test_prepared = preprocessor.transform(X_test[features])\n# make predictions\npreds = grd_search_gb.predict(X_test_prepared)\n\n# print(classification_report(preds, y_test))","metadata":{"execution":{"iopub.status.busy":"2022-08-04T20:32:27.822527Z","iopub.execute_input":"2022-08-04T20:32:27.823213Z","iopub.status.idle":"2022-08-04T20:32:27.842622Z","shell.execute_reply.started":"2022-08-04T20:32:27.823175Z","shell.execute_reply":"2022-08-04T20:32:27.841645Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# tn, fp, fn, tp = confusion_matrix(preds, y_test).ravel()\n# print(tn)\n# print(fp)\n# print(fn)\n# print(tp)\n# print('precision_class1:\\t{0:.2f}'.format(tp / (tp + fp)))\n# print('recall_class1:\\t\\t{0:.2f}'.format(tp / (tp + fn)))\n# print('precision_class0:\\t{0:.2f}'.format(tn / (tn + fn)))\n# print('recall_class0:\\t\\t{0:.2f}'.format(tn / (tn + fp)))\n\n# print(sns.heatmap(confusion_matrix(preds, y_test), annot=True, cmap='Blues'))","metadata":{"execution":{"iopub.status.busy":"2022-08-04T20:32:38.790752Z","iopub.execute_input":"2022-08-04T20:32:38.791391Z","iopub.status.idle":"2022-08-04T20:32:38.796301Z","shell.execute_reply.started":"2022-08-04T20:32:38.791339Z","shell.execute_reply":"2022-08-04T20:32:38.795196Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"output = pd.DataFrame({'PassengerId': test_data.PassengerId, 'Survived': preds})\noutput.to_csv('my_submission.csv', index=False)\nprint(\"Your submission was successfully saved!\")","metadata":{"execution":{"iopub.status.busy":"2022-08-04T20:33:22.751900Z","iopub.execute_input":"2022-08-04T20:33:22.752612Z","iopub.status.idle":"2022-08-04T20:33:23.075719Z","shell.execute_reply.started":"2022-08-04T20:33:22.752547Z","shell.execute_reply":"2022-08-04T20:33:23.074895Z"},"trusted":true},"execution_count":null,"outputs":[]}]}