{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport seaborn as sns\nfrom matplotlib import style\nfrom matplotlib import pyplot as plt\nstyle.use('bmh')\n%matplotlib inline\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-02T20:27:51.729105Z","iopub.execute_input":"2022-07-02T20:27:51.729489Z","iopub.status.idle":"2022-07-02T20:27:51.744129Z","shell.execute_reply.started":"2022-07-02T20:27:51.729457Z","shell.execute_reply":"2022-07-02T20:27:51.743027Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"![titanic](https://static.timesofisrael.com/atlantajewishtimes/uploads/2022/03/DT6RD9.jpg) ![]()","metadata":{}},{"cell_type":"markdown","source":"## Table of contents\n* [Chapter 1 Introduction](#chapter1)\n    * [Section 1.1 The Data](#subsection1)\n    * [Section 1.2 Exploring Data](#subsection2)\n* [Chapter 2 - Cleaning data](#chapter2)\n    * [Section 2.1 Removed missing data](#subsection2.1)\n    * [Section 2.2 Encoding](#subsection2.2)\n    * [Section 2.3 Scaling](#subsection2.3)\n    * [Section 2.4 Removing redundant features](#subsection2.4)\n* [Chapter 3 - Modelling](#chapter3)\n    * [Section 3.1 Training and fitting classifier](#subsection3.1)\n    * [Section 3.1 Brief evaluation](#subsection3.2)\n","metadata":{}},{"cell_type":"markdown","source":"## 1. Introduction  <a class=\"anchor\"  id=\"chapter1\"></a>","metadata":{}},{"cell_type":"markdown","source":"<p style=\"padding: 10px;\n              color:white;\">\n<div style=\"color:white;\n           display:fill;\n           border-radius:10px;\n           background-color:#141855;\n           font-size:210%;\n           font-family:Helvetica;\n           letter-spacing:0.5px\">\n<center>\"Let the Truth be known, no ship is unsinkable. The bigger the ship, the easier it is to sink her.\" - Thomas Andrews</center>\n    \n</p>\n</div>\n\nThe sinking of the Titanic is one of the most infamous shipwrecks in history.\n\nOn April 15, 1912, during her maiden voyage, the widely considered “unsinkable” RMS Titanic sank after colliding with an iceberg. Unfortunately, there weren’t enough lifeboats for everyone onboard, resulting in the death of 1502 out of 2224 passengers and crew.\n\nWhile there was some element of luck involved in surviving, it seems some groups of people were more likely to survive than others.","metadata":{}},{"cell_type":"code","source":"#load in the data\n# merging the files can save a few lines of code\ntrain_df = pd.read_csv('/kaggle/input/titanic/train.csv')\ntest_df = pd.read_csv('/kaggle/input/titanic/test.csv')\ntest_df2 = pd.read_csv('/kaggle/input/titanic/test.csv')\ndf = pd.concat([train_df,test_df])\ntraindex = train_df.index\ntestdex = test_df.index","metadata":{"execution":{"iopub.status.busy":"2022-07-02T20:27:51.859685Z","iopub.execute_input":"2022-07-02T20:27:51.860688Z","iopub.status.idle":"2022-07-02T20:27:51.885915Z","shell.execute_reply.started":"2022-07-02T20:27:51.860652Z","shell.execute_reply":"2022-07-02T20:27:51.88457Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"  ## 1.1. View data <a class=\"anchor\"  id=\"subsection1\"></a>","metadata":{}},{"cell_type":"code","source":"train_df","metadata":{"execution":{"iopub.status.busy":"2022-07-02T20:27:52.134356Z","iopub.execute_input":"2022-07-02T20:27:52.134805Z","iopub.status.idle":"2022-07-02T20:27:52.165404Z","shell.execute_reply.started":"2022-07-02T20:27:52.13477Z","shell.execute_reply":"2022-07-02T20:27:52.164417Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df","metadata":{"execution":{"iopub.status.busy":"2022-07-02T20:27:52.356972Z","iopub.execute_input":"2022-07-02T20:27:52.357656Z","iopub.status.idle":"2022-07-02T20:27:52.385916Z","shell.execute_reply.started":"2022-07-02T20:27:52.357618Z","shell.execute_reply":"2022-07-02T20:27:52.384703Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df","metadata":{"execution":{"iopub.status.busy":"2022-07-02T20:27:52.546945Z","iopub.execute_input":"2022-07-02T20:27:52.547373Z","iopub.status.idle":"2022-07-02T20:27:52.574499Z","shell.execute_reply.started":"2022-07-02T20:27:52.547329Z","shell.execute_reply":"2022-07-02T20:27:52.573165Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"  ## 1.2. Explore data <a class=\"anchor\"  id=\"subsection2\"></a>","metadata":{}},{"cell_type":"code","source":"#missing data \nplt.figure(figsize=(16,5))\nsns.heatmap(train_df.isnull(), cmap='rocket')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-02T20:27:52.599881Z","iopub.execute_input":"2022-07-02T20:27:52.60028Z","iopub.status.idle":"2022-07-02T20:27:53.104493Z","shell.execute_reply.started":"2022-07-02T20:27:52.600248Z","shell.execute_reply":"2022-07-02T20:27:53.103537Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# do larger families suffer more?\naxes = sns.factorplot('SibSp','Survived', \n                      data=train_df, aspect = 2.5, )","metadata":{"execution":{"iopub.status.busy":"2022-07-02T20:27:53.107031Z","iopub.execute_input":"2022-07-02T20:27:53.107784Z","iopub.status.idle":"2022-07-02T20:27:53.738162Z","shell.execute_reply.started":"2022-07-02T20:27:53.107745Z","shell.execute_reply":"2022-07-02T20:27:53.736787Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#suffice to say the poor do\ng = sns.catplot(y=\"Survived\", col=\"Pclass\",\n                 data=df, saturation=.5,kind=\"bar\",ci=None, aspect=.6)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-02T20:27:53.739988Z","iopub.execute_input":"2022-07-02T20:27:53.740479Z","iopub.status.idle":"2022-07-02T20:27:54.220194Z","shell.execute_reply.started":"2022-07-02T20:27:53.740429Z","shell.execute_reply":"2022-07-02T20:27:54.218791Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#no comment\nsns.barplot(data=train_df, x='Sex', y='Survived')","metadata":{"execution":{"iopub.status.busy":"2022-07-02T20:27:54.222912Z","iopub.execute_input":"2022-07-02T20:27:54.223261Z","iopub.status.idle":"2022-07-02T20:27:54.514041Z","shell.execute_reply.started":"2022-07-02T20:27:54.223227Z","shell.execute_reply":"2022-07-02T20:27:54.512767Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 2. Cleaning  <a class=\"anchor\"  id=\"chapter2\"></a>","metadata":{}},{"cell_type":"code","source":"#replace numeric features with medians and cat with mode\ndf.Age.fillna(df.Age.median(),inplace=True)\ndf.Cabin.fillna(df.Cabin.mode()[0],inplace=True)\ndf.Fare.fillna(df.Fare.median(),inplace=True)\ndf.Embarked.fillna(df.Embarked.mode()[0],inplace=True)\ndf.Age.fillna(df.Age.median(),inplace=True)\ndf.Cabin.fillna(df.Cabin.mode()[0],inplace=True)\ndf.Fare.fillna(df.Fare.median(),inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-02T20:47:57.996934Z","iopub.execute_input":"2022-07-02T20:47:57.997347Z","iopub.status.idle":"2022-07-02T20:47:58.025484Z","shell.execute_reply.started":"2022-07-02T20:47:57.997315Z","shell.execute_reply":"2022-07-02T20:47:58.02339Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#changing cabin gibberish to cabin type\ndf_cabin = list(df.Cabin)\ncabin_letters = []\nfor i in range(len(df_cabin)):\n    cabin_letters.append(df_cabin[i][0])\ncabin_letters_series=pd.Series(cabin_letters)\ndf['CabinLetter'] = cabin_letters_series\nplt.figure(figsize=(20, 7))\nsns.barplot(data=df, x='CabinLetter', y='Survived', palette=\"Blues_d\")","metadata":{"execution":{"iopub.status.busy":"2022-07-02T20:27:54.536416Z","iopub.execute_input":"2022-07-02T20:27:54.536809Z","iopub.status.idle":"2022-07-02T20:27:55.094077Z","shell.execute_reply.started":"2022-07-02T20:27:54.536768Z","shell.execute_reply":"2022-07-02T20:27:55.092941Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# according to almost everyone on this site, models get mislead by sex and age, \n# without considering title\ndf.Title = 0\ndf['Title']=df.Name.str.extract('([A-Za-z]+)\\.')\ndf['Title'].replace(['Mlle','Mme','Ms','Dr','Major','Lady','Countess','Jonkheer','Col',\n                         'Rev','Capt','Sir','Don'],['Miss','Miss','Miss','Mr','Mr','Mrs','Mrs','Other','Other','Other','Mr','Mr','Mr'],inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-02T20:27:55.095564Z","iopub.execute_input":"2022-07-02T20:27:55.095937Z","iopub.status.idle":"2022-07-02T20:27:55.114039Z","shell.execute_reply.started":"2022-07-02T20:27:55.095895Z","shell.execute_reply":"2022-07-02T20:27:55.112145Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(20, 7))\nsns.barplot(data=df, x='Title', y='Survived', palette=\"viridis\")","metadata":{"execution":{"iopub.status.busy":"2022-07-02T20:27:55.11558Z","iopub.execute_input":"2022-07-02T20:27:55.115973Z","iopub.status.idle":"2022-07-02T20:27:55.544647Z","shell.execute_reply.started":"2022-07-02T20:27:55.11594Z","shell.execute_reply":"2022-07-02T20:27:55.543388Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# don't need this anymore\ndf.drop(columns=['Name'],inplace = True)\n# df.drop(columns=['Ticket'],inplace = True)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-02T20:27:55.548318Z","iopub.execute_input":"2022-07-02T20:27:55.548794Z","iopub.status.idle":"2022-07-02T20:27:55.556103Z","shell.execute_reply.started":"2022-07-02T20:27:55.548748Z","shell.execute_reply":"2022-07-02T20:27:55.555109Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# plotting distributions\nf, (ax1, ax2) = plt.subplots(1, 2, figsize=(21, 8), sharex=True)\nax1.title.set_text('Age and Sex Distribution')\nax2.title.set_text('Age and Survived Distribution')\nsns.kdeplot(data=train_df, x=\"Age\", hue=\"Sex\", fill=True, common_norm=False, palette=\"viridis\", alpha=.5, linewidth=0, ax=ax1)\nsns.kdeplot(data=train_df, x=\"Age\", hue=\"Survived\", fill=True, common_norm=False, palette=\"rocket\", alpha=.5, linewidth=0, ax=ax2)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-02T20:27:55.560475Z","iopub.execute_input":"2022-07-02T20:27:55.561542Z","iopub.status.idle":"2022-07-02T20:27:56.054269Z","shell.execute_reply.started":"2022-07-02T20:27:55.561503Z","shell.execute_reply":"2022-07-02T20:27:56.053391Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2022-07-02T20:27:56.05564Z","iopub.execute_input":"2022-07-02T20:27:56.056177Z","iopub.status.idle":"2022-07-02T20:27:56.065481Z","shell.execute_reply.started":"2022-07-02T20:27:56.056145Z","shell.execute_reply":"2022-07-02T20:27:56.064399Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":" ## 2.1. Missing data <a class=\"anchor\"  id=\"subsection2.1\"></a>","metadata":{}},{"cell_type":"code","source":"test_df.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2022-07-02T20:27:56.066604Z","iopub.execute_input":"2022-07-02T20:27:56.067318Z","iopub.status.idle":"2022-07-02T20:27:56.077806Z","shell.execute_reply.started":"2022-07-02T20:27:56.067286Z","shell.execute_reply":"2022-07-02T20:27:56.076575Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# the data is not disbalanced enough to merit resampling\nax = sns.countplot(x=\"Survived\", data=train_df)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-02T20:27:56.079219Z","iopub.execute_input":"2022-07-02T20:27:56.07969Z","iopub.status.idle":"2022-07-02T20:27:56.257333Z","shell.execute_reply.started":"2022-07-02T20:27:56.079644Z","shell.execute_reply":"2022-07-02T20:27:56.256025Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# quick correlations\nsns.heatmap(df.corr())","metadata":{"execution":{"iopub.status.busy":"2022-07-02T20:27:56.274684Z","iopub.execute_input":"2022-07-02T20:27:56.275126Z","iopub.status.idle":"2022-07-02T20:27:56.639861Z","shell.execute_reply.started":"2022-07-02T20:27:56.275091Z","shell.execute_reply":"2022-07-02T20:27:56.638906Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# these seem pretty random\ndf.drop(columns=['Cabin','CabinLetter'])","metadata":{"execution":{"iopub.status.busy":"2022-07-02T20:27:56.687837Z","iopub.execute_input":"2022-07-02T20:27:56.689166Z","iopub.status.idle":"2022-07-02T20:27:56.717388Z","shell.execute_reply.started":"2022-07-02T20:27:56.689126Z","shell.execute_reply":"2022-07-02T20:27:56.71617Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"  ## 2.2. Encoding <a class=\"anchor\"  id=\"subsection2.2\"></a>","metadata":{}},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"# one single feature for family size \ndf['FamilySize'] = df['SibSp'] + df['Parch'] + 1\ndf['IsAlone'] = 0\ndf.loc[df['FamilySize'] == 1, 'IsAlone'] = 1\ndf['Sex'] = df['Sex'].map( {'female': 1, 'male': 0} ).astype(int)\n\n","metadata":{"execution":{"iopub.status.busy":"2022-07-02T20:27:56.752517Z","iopub.execute_input":"2022-07-02T20:27:56.752864Z","iopub.status.idle":"2022-07-02T20:27:56.765838Z","shell.execute_reply.started":"2022-07-02T20:27:56.752832Z","shell.execute_reply":"2022-07-02T20:27:56.764452Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#one-hot encode\ndf = pd.get_dummies(df,columns=['Embarked'],drop_first=True)\ndf = pd.get_dummies(df,columns=['Title'],drop_first=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-02T20:27:56.767157Z","iopub.execute_input":"2022-07-02T20:27:56.767469Z","iopub.status.idle":"2022-07-02T20:27:56.786624Z","shell.execute_reply.started":"2022-07-02T20:27:56.767437Z","shell.execute_reply":"2022-07-02T20:27:56.785548Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# check for outliers\ndf.boxplot(column=['Age', 'Fare'])  ","metadata":{"execution":{"iopub.status.busy":"2022-07-02T20:27:56.788008Z","iopub.execute_input":"2022-07-02T20:27:56.788395Z","iopub.status.idle":"2022-07-02T20:27:56.998532Z","shell.execute_reply.started":"2022-07-02T20:27:56.788362Z","shell.execute_reply":"2022-07-02T20:27:56.997204Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"  ## 2.3. Scaling numerical data <a class=\"anchor\"  id=\"subsection2.3\"></a>","metadata":{}},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"# normalise the data and disregard above boxplot due to poorer performance (yes this was ab tested)\ndf.drop(columns=['SibSp','Ticket','Cabin','CabinLetter','PassengerId','Parch','Sex'])\nfrom sklearn.preprocessing import RobustScaler, StandardScaler\nscaler = StandardScaler()\ndf[['Age', 'Fare']] = scaler.fit_transform(df[['Age', 'Fare']])","metadata":{"execution":{"iopub.status.busy":"2022-07-02T20:27:57.000188Z","iopub.execute_input":"2022-07-02T20:27:57.000517Z","iopub.status.idle":"2022-07-02T20:27:57.014239Z","shell.execute_reply.started":"2022-07-02T20:27:57.000484Z","shell.execute_reply":"2022-07-02T20:27:57.012791Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# high scoring users say this matters\ndf['ticket_start'] = df.Ticket.apply(lambda x: x[:2])\ndf['ticket_length'] = df.Ticket.apply(lambda x: len(x))\ndf = pd.get_dummies(df,columns=['ticket_start','ticket_length'],drop_first=True)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-02T20:27:57.015672Z","iopub.execute_input":"2022-07-02T20:27:57.016628Z","iopub.status.idle":"2022-07-02T20:27:57.034596Z","shell.execute_reply.started":"2022-07-02T20:27:57.01658Z","shell.execute_reply":"2022-07-02T20:27:57.033445Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"  ## 2.4. Remove redundant features <a class=\"anchor\"  id=\"subsection2.4\"></a>","metadata":{}},{"cell_type":"code","source":"df.drop(columns=['Sex','Age','SibSp','Parch','Ticket','Cabin','CabinLetter','PassengerId'],inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-02T20:27:57.036157Z","iopub.execute_input":"2022-07-02T20:27:57.036851Z","iopub.status.idle":"2022-07-02T20:27:57.045751Z","shell.execute_reply.started":"2022-07-02T20:27:57.036816Z","shell.execute_reply":"2022-07-02T20:27:57.044619Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"markdown","source":"## 3. Modelling  <a class=\"anchor\"  id=\"chapter3\"></a>","metadata":{}},{"cell_type":"code","source":"#before scaling these, need to split\nfrom sklearn.model_selection import train_test_split\ntrain_df = df.iloc[traindex,:]\ntrain_df.Survived = train_df.Survived.astype(int)\ntest_df = df.iloc[891:]\nX = train_df.loc[:,train_df.columns!='Survived']\ny = train_df.loc[:,train_df.columns=='Survived']\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.3, random_state=42)","metadata":{"execution":{"iopub.status.busy":"2022-07-02T20:27:57.047274Z","iopub.execute_input":"2022-07-02T20:27:57.047614Z","iopub.status.idle":"2022-07-02T20:27:57.061264Z","shell.execute_reply.started":"2022-07-02T20:27:57.047583Z","shell.execute_reply":"2022-07-02T20:27:57.060029Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df.drop(columns=['Survived'],inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-02T20:27:57.098628Z","iopub.execute_input":"2022-07-02T20:27:57.099018Z","iopub.status.idle":"2022-07-02T20:27:57.106312Z","shell.execute_reply.started":"2022-07-02T20:27:57.098985Z","shell.execute_reply":"2022-07-02T20:27:57.10541Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 3.1 Classifier fitting  <a class=\"anchor\"  id=\"subsection3.1\"></a>","metadata":{}},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"from sklearn.ensemble import GradientBoostingClassifier\nfrom sklearn.model_selection import GridSearchCV\n\nclf = GradientBoostingClassifier(n_estimators=100, learning_rate=0.01,\n                                 max_depth=1, random_state=0).fit(X_train, y_train)\nclf.score(X_test, y_test)","metadata":{"execution":{"iopub.status.busy":"2022-07-02T20:27:57.107765Z","iopub.execute_input":"2022-07-02T20:27:57.108062Z","iopub.status.idle":"2022-07-02T20:27:57.208431Z","shell.execute_reply.started":"2022-07-02T20:27:57.108031Z","shell.execute_reply":"2022-07-02T20:27:57.207001Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# hyperparameter tuning can often lead to a more effective model but can take too long\n# try it out for yourself\n\n# params= {\n#     'n_estimators': [100,200,500,1000],\n#     'learning_rate': [5,1,0.1,0.001,0.0001],\n#     'max_depth': [5,3,1]\n# }\n\n\n# clf = GridSearchCV(\n#     estimator=clf,\n#     param_grid=params,    \n#     cv=10,            \n#     scoring='accuracy',       \n#     n_jobs=-1,        \n#     verbose=10        \n# )\n# clf.fit(X_train, y_train)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-02T20:27:57.210324Z","iopub.execute_input":"2022-07-02T20:27:57.21107Z","iopub.status.idle":"2022-07-02T20:27:57.216312Z","shell.execute_reply.started":"2022-07-02T20:27:57.211034Z","shell.execute_reply":"2022-07-02T20:27:57.215274Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# print(clf.best_params_)","metadata":{"execution":{"iopub.status.busy":"2022-07-02T20:27:57.2176Z","iopub.execute_input":"2022-07-02T20:27:57.218529Z","iopub.status.idle":"2022-07-02T20:27:57.230757Z","shell.execute_reply.started":"2022-07-02T20:27:57.218496Z","shell.execute_reply":"2022-07-02T20:27:57.229389Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# the choice for this task is logistic regression, random forest or something akin to xgb\n# it appears to be the case that most users go with random forest! lets try something else\nmy_model = GradientBoostingClassifier(random_state=100, n_estimators=200, learning_rate=0.023, max_features=3)\nmy_model.fit(X_train,y_train)","metadata":{"execution":{"iopub.status.busy":"2022-07-02T20:27:57.232788Z","iopub.execute_input":"2022-07-02T20:27:57.234039Z","iopub.status.idle":"2022-07-02T20:27:57.403424Z","shell.execute_reply.started":"2022-07-02T20:27:57.233973Z","shell.execute_reply":"2022-07-02T20:27:57.402113Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#just to check...\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.ensemble import RandomForestClassifier\nclf = LogisticRegression(random_state=0).fit(X_train, y_train)\nclf = RandomForestClassifier(random_state=100, n_estimators=1000, min_samples_split= 6, min_samples_leaf=2, max_depth=10).fit(X_train,y_train)","metadata":{"execution":{"iopub.status.busy":"2022-07-02T20:29:32.356905Z","iopub.execute_input":"2022-07-02T20:29:32.357306Z","iopub.status.idle":"2022-07-02T20:29:34.514827Z","shell.execute_reply.started":"2022-07-02T20:29:32.357274Z","shell.execute_reply":"2022-07-02T20:29:34.513501Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 3.2 Brief evaluation  <a class=\"anchor\"  id=\"subsection3.2\"></a>","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import cross_val_score\npreds = clf.predict(X_test)\ncross_valid = cross_val_score(clf,\n                              X_train, y_train, cv=10, scoring='accuracy').mean()\ncross_valid","metadata":{"execution":{"iopub.status.busy":"2022-07-02T20:34:57.035973Z","iopub.execute_input":"2022-07-02T20:34:57.03639Z","iopub.status.idle":"2022-07-02T20:35:19.421625Z","shell.execute_reply.started":"2022-07-02T20:34:57.036354Z","shell.execute_reply":"2022-07-02T20:35:19.420269Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# training data provided by kaggle ROC curve using ensemble model\nimport scikitplot as skplt\nmodel = my_model\nmodel.fit(X_train, y_train)\ny_probas = model.predict_proba(X_test)\nskplt.metrics.plot_roc(y_test, y_probas)\nplt.rcParams[\"figure.figsize\"] = (15,10)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-02T21:11:59.606031Z","iopub.execute_input":"2022-07-02T21:11:59.606438Z","iopub.status.idle":"2022-07-02T21:12:00.09167Z","shell.execute_reply.started":"2022-07-02T21:11:59.606406Z","shell.execute_reply":"2022-07-02T21:12:00.09019Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#create predictions\npredicted = my_model.predict(test_df)","metadata":{"execution":{"iopub.status.busy":"2022-07-02T20:29:34.516683Z","iopub.execute_input":"2022-07-02T20:29:34.517059Z","iopub.status.idle":"2022-07-02T20:29:34.733412Z","shell.execute_reply.started":"2022-07-02T20:29:34.517027Z","shell.execute_reply":"2022-07-02T20:29:34.732175Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#submit\nsubmission = pd.DataFrame({'PassengerId':test_df2['PassengerId'],'Survived':predicted})\nsubmission.to_csv(\"submission.csv\", header=True,index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-02T20:27:57.539273Z","iopub.status.idle":"2022-07-02T20:27:57.540656Z","shell.execute_reply.started":"2022-07-02T20:27:57.54032Z","shell.execute_reply":"2022-07-02T20:27:57.540352Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission","metadata":{"execution":{"iopub.status.busy":"2022-07-02T20:27:57.542578Z","iopub.status.idle":"2022-07-02T20:27:57.543844Z","shell.execute_reply.started":"2022-07-02T20:27:57.543444Z","shell.execute_reply":"2022-07-02T20:27:57.543481Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.status.busy":"2022-07-02T21:07:08.55954Z","iopub.execute_input":"2022-07-02T21:07:08.561059Z","iopub.status.idle":"2022-07-02T21:07:08.571399Z","shell.execute_reply.started":"2022-07-02T21:07:08.560998Z","shell.execute_reply":"2022-07-02T21:07:08.569809Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}