{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nfrom sklearn.model_selection import KFold\nfrom sklearn.metrics import accuracy_score\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.compose import ColumnTransformer\nfrom sklearn.ensemble import GradientBoostingClassifier, RandomForestClassifier\nfrom sklearn.linear_model import Perceptron\nfrom catboost import CatBoostClassifier\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.model_selection import cross_val_score, GridSearchCV\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.preprocessing import LabelBinarizer, OneHotEncoder, StandardScaler\nfrom sklearn import set_config\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\nimport matplotlib.pyplot as plt \nimport seaborn as sns\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-23T15:06:37.832120Z","iopub.execute_input":"2022-07-23T15:06:37.832779Z","iopub.status.idle":"2022-07-23T15:06:37.849046Z","shell.execute_reply.started":"2022-07-23T15:06:37.832726Z","shell.execute_reply":"2022-07-23T15:06:37.848002Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"This function allows us to create a submission easily","metadata":{}},{"cell_type":"code","source":"def make_submission():\n    submission = titanic_test.copy()\n    #display(titanic_test)\n    Y_test_pred = model.predict(X_test)\n    display(X_test)\n    submission['Survived'] = Y_test_pred\n    submission.drop(submission.iloc[:, 1:-1], inplace=True, axis=1)\n    submission = submission.set_index('PassengerId')\n    display(submission)\n    submission.to_csv('/kaggle/working/submissionZ.csv')","metadata":{"execution":{"iopub.status.busy":"2022-07-23T15:06:37.891210Z","iopub.execute_input":"2022-07-23T15:06:37.892353Z","iopub.status.idle":"2022-07-23T15:06:37.900674Z","shell.execute_reply.started":"2022-07-23T15:06:37.892300Z","shell.execute_reply":"2022-07-23T15:06:37.899313Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Titanic - Machine Learning from Disaster","metadata":{}},{"cell_type":"markdown","source":"Is it possible to predict whether someone will survive a given disaster? That is the question we'll try to answer.\n\nHere, we'll be using the example of the Titanic and its passengers. ","metadata":{}},{"cell_type":"markdown","source":"## Let's load the data","metadata":{}},{"cell_type":"markdown","source":"# Explanations for the columns of our dataset:\n\n**Variable**\tDefinition\tKey\n**survival**\tSurvival\t0 = No, 1 = Yes\n**pclass**\tTicket class\t1 = 1st, 2 = 2nd, 3 = 3rd\n**sex**\tSex\t\n**Age**\tAge in years\t\n**sibsp**\t# of siblings / spouses aboard the Titanic\t\n**parch**\t# of parents / children aboard the Titanic\t\n**ticket**\tTicket number\t\n**fare**\tPassenger fare\t\n**cabin**\tCabin number\t\n**embarked**\tPort of Embarkation\tC = Cherbourg, Q = Queenstown, S = Southampton\n\n**pclass**: A proxy for socio-economic status (SES)\n1st = Upper\n2nd = Middle\n3rd = Lower\n\n**age**: Age is fractional if less than 1. If the age is estimated, is it in the form of xx.5\n\n**sibsp**: The dataset defines family relations in this way...\n**Sibling** = brother, sister, stepbrother, stepsister\n**Spouse** = husband, wife (mistresses and fiancés were ignored)\n\n**parch**: The dataset defines family relations in this way...\n**Parent** = mother, father\n**Child** = daughter, son, stepdaughter, stepson\nSome children travelled only with a nanny, therefore parch=0 for them.","metadata":{}},{"cell_type":"markdown","source":"# Data Exploration","metadata":{}},{"cell_type":"code","source":"titanic_train = pd.read_csv('/kaggle/input/titanic/train.csv')\ntitanic_test = pd.read_csv('/kaggle/input/titanic/test.csv')\nprint(titanic_train.shape)\ndisplay(titanic_train)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-23T15:06:37.967288Z","iopub.execute_input":"2022-07-23T15:06:37.967676Z","iopub.status.idle":"2022-07-23T15:06:38.009326Z","shell.execute_reply.started":"2022-07-23T15:06:37.967646Z","shell.execute_reply":"2022-07-23T15:06:38.008053Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We can see that there are 11 features and we are trying to predict the survived column. \n\nSome columns like Age and Cabin seem to cointain NaN values which is a problem. to fill values when it is needed.\n\nThe ticket number doesn't give much information since we already have the fare and the cabin. Thus, we can delete this column.\nSame goes for the Name column\n\nWe'll use one-hot vectors for Sex and Embarked \n\n\n\n","metadata":{}},{"cell_type":"code","source":"sns.catplot(x=\"Survived\", kind=\"count\",palette=\"magma\", data=titanic_train, height = 6)\nplt.title(\"Survived (binary: yes or no)\")","metadata":{"execution":{"iopub.status.busy":"2022-07-23T15:06:38.029369Z","iopub.execute_input":"2022-07-23T15:06:38.029743Z","iopub.status.idle":"2022-07-23T15:06:38.427625Z","shell.execute_reply.started":"2022-07-23T15:06:38.029712Z","shell.execute_reply":"2022-07-23T15:06:38.426330Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We can see that not everyone died, almost 300 people survived. Thus the classes aren't very imbalanced.","metadata":{"execution":{"iopub.status.busy":"2022-07-23T12:33:37.596969Z","iopub.execute_input":"2022-07-23T12:33:37.597534Z","iopub.status.idle":"2022-07-23T12:33:37.606099Z","shell.execute_reply.started":"2022-07-23T12:33:37.597501Z","shell.execute_reply":"2022-07-23T12:33:37.603165Z"}}},{"cell_type":"markdown","source":"# Data Preparation","metadata":{}},{"cell_type":"markdown","source":"We will build a pipeline so that we can clean our data easily.","metadata":{}},{"cell_type":"code","source":"#constants\nfeatures = ['Pclass', 'Name', 'Sex', 'Age', 'SibSp', 'Parch', 'Ticket', 'Fare', 'Cabin', 'Embarked']\ntarget = 'Survived'\n\ndef pipeline(numerical_imputer, numerical_scaler, numerical_features, categorical_imputer, categorical_encoder, categorical_features, estimator):\n    numerical_transformer = Pipeline(\n        steps=[\n            (\"numerical_imputer\", numerical_imputer),\n            (\"numerical_scaler\", numerical_scaler),\n        ]\n    )\n    categorical_transformer = Pipeline(\n        steps=[\n            (\"categorical_imputer\", categorical_imputer),\n            (\"categorical_encoder\", categorical_encoder),\n        ]\n    )\n    preprocessor = ColumnTransformer(\n        transformers=[\n            (\"num\", numerical_transformer, numerical_features),\n            (\"cat\", categorical_transformer, categorical_features),\n        ]\n    )\n    clf = Pipeline( \n        steps=[(\"preprocessor\", preprocessor), (\"classifier\", estimator)]\n    )\n    return clf","metadata":{"execution":{"iopub.status.busy":"2022-07-23T15:06:38.431678Z","iopub.execute_input":"2022-07-23T15:06:38.432515Z","iopub.status.idle":"2022-07-23T15:06:38.443504Z","shell.execute_reply.started":"2022-07-23T15:06:38.432467Z","shell.execute_reply":"2022-07-23T15:06:38.442224Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"random_state = 0\n\n\nnumerical_imputer = SimpleImputer(strategy='mean')\nnumerical_scaler = StandardScaler()\nnumerical_features = ['Age', 'Fare']\ncategorical_imputer = SimpleImputer(strategy='most_frequent')\ncategorical_encoder = OneHotEncoder(handle_unknown='ignore')\ncategorical_features = ['Pclass'\n            , 'Sex'\t\n            , 'SibSp'\n            , 'Parch'\n            , 'Ticket'\n            , 'Embarked'\n           ]\n","metadata":{"execution":{"iopub.status.busy":"2022-07-23T15:06:38.445314Z","iopub.execute_input":"2022-07-23T15:06:38.445770Z","iopub.status.idle":"2022-07-23T15:06:38.455812Z","shell.execute_reply.started":"2022-07-23T15:06:38.445724Z","shell.execute_reply":"2022-07-23T15:06:38.455005Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train = titanic_train[numerical_features + categorical_features]\nX_test = titanic_test[numerical_features + categorical_features]\ny_train = titanic_train[target]","metadata":{"execution":{"iopub.status.busy":"2022-07-23T15:06:38.459538Z","iopub.execute_input":"2022-07-23T15:06:38.460678Z","iopub.status.idle":"2022-07-23T15:06:38.469447Z","shell.execute_reply.started":"2022-07-23T15:06:38.460630Z","shell.execute_reply":"2022-07-23T15:06:38.468382Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-23T15:06:38.471421Z","iopub.execute_input":"2022-07-23T15:06:38.472288Z","iopub.status.idle":"2022-07-23T15:06:38.490215Z","shell.execute_reply.started":"2022-07-23T15:06:38.472239Z","shell.execute_reply":"2022-07-23T15:06:38.489045Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_t, X_val, y_t, y_val = train_test_split(X_train, y_train, test_size=0.2, random_state=42)","metadata":{"execution":{"iopub.status.busy":"2022-07-23T15:06:38.492626Z","iopub.execute_input":"2022-07-23T15:06:38.493463Z","iopub.status.idle":"2022-07-23T15:06:38.502288Z","shell.execute_reply.started":"2022-07-23T15:06:38.493418Z","shell.execute_reply":"2022-07-23T15:06:38.500446Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Our baseline will be the score of a Perceptron algorithm","metadata":{}},{"cell_type":"code","source":"estimator = Perceptron(random_state = random_state)","metadata":{"execution":{"iopub.status.busy":"2022-07-23T15:06:38.503690Z","iopub.execute_input":"2022-07-23T15:06:38.504115Z","iopub.status.idle":"2022-07-23T15:06:38.510964Z","shell.execute_reply.started":"2022-07-23T15:06:38.504059Z","shell.execute_reply":"2022-07-23T15:06:38.510158Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = pipeline(numerical_imputer = numerical_imputer\n                 , numerical_scaler = numerical_scaler\n                 , numerical_features = numerical_features  \n                 , categorical_imputer = categorical_imputer\n                 , categorical_encoder = categorical_encoder\n                 , categorical_features = categorical_features\n                 , estimator = estimator)","metadata":{"execution":{"iopub.status.busy":"2022-07-23T15:06:38.513328Z","iopub.execute_input":"2022-07-23T15:06:38.513646Z","iopub.status.idle":"2022-07-23T15:06:38.525800Z","shell.execute_reply.started":"2022-07-23T15:06:38.513617Z","shell.execute_reply":"2022-07-23T15:06:38.524955Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"set_config(display=\"diagram\")\nmodel","metadata":{"execution":{"iopub.status.busy":"2022-07-23T15:06:38.580087Z","iopub.execute_input":"2022-07-23T15:06:38.580759Z","iopub.status.idle":"2022-07-23T15:06:38.656641Z","shell.execute_reply.started":"2022-07-23T15:06:38.580709Z","shell.execute_reply":"2022-07-23T15:06:38.655629Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"scores = cross_val_score(model, X_t, y_t, scoring=\"accuracy\", cv=10)\nprint(\"Average CV score:\", scores.mean())\nprint(\"CV score standard deviation: \", scores.std())\n","metadata":{"execution":{"iopub.status.busy":"2022-07-23T15:06:38.697526Z","iopub.execute_input":"2022-07-23T15:06:38.697931Z","iopub.status.idle":"2022-07-23T15:06:38.991299Z","shell.execute_reply.started":"2022-07-23T15:06:38.697895Z","shell.execute_reply":"2022-07-23T15:06:38.990175Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.fit(X_t, y_t)\nY_val_pred = model.predict(X_val)\n\nfrom sklearn.metrics import ConfusionMatrixDisplay\n\nConfusionMatrixDisplay.from_predictions(y_val, Y_val_pred)\nplt.title(\"Confusion matrix for training set\")\nplt.show()\n\nprint(accuracy_score(y_val, Y_val_pred))","metadata":{"execution":{"iopub.status.busy":"2022-07-23T15:06:38.993463Z","iopub.execute_input":"2022-07-23T15:06:38.994504Z","iopub.status.idle":"2022-07-23T15:06:39.216770Z","shell.execute_reply.started":"2022-07-23T15:06:38.994447Z","shell.execute_reply":"2022-07-23T15:06:39.215501Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Let's try another model using, XGBoost :","metadata":{}},{"cell_type":"code","source":"estimator = GradientBoostingClassifier(random_state = random_state)\nmodel = pipeline(numerical_imputer = numerical_imputer\n                 , numerical_scaler = numerical_scaler\n                 , numerical_features = numerical_features  \n                 , categorical_imputer = categorical_imputer\n                 , categorical_encoder = categorical_encoder\n                 , categorical_features = categorical_features\n                 , estimator = estimator)","metadata":{"execution":{"iopub.status.busy":"2022-07-23T15:06:39.218386Z","iopub.execute_input":"2022-07-23T15:06:39.219401Z","iopub.status.idle":"2022-07-23T15:06:39.226766Z","shell.execute_reply.started":"2022-07-23T15:06:39.219350Z","shell.execute_reply":"2022-07-23T15:06:39.225679Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"scores = cross_val_score(model, X_t, y_t, scoring=\"accuracy\", cv=10)\nprint(\"Average CV score:\", scores.mean())\nprint(\"CV score standard deviation: \", scores.std())\n","metadata":{"execution":{"iopub.status.busy":"2022-07-23T15:06:39.229276Z","iopub.execute_input":"2022-07-23T15:06:39.229966Z","iopub.status.idle":"2022-07-23T15:06:40.878348Z","shell.execute_reply.started":"2022-07-23T15:06:39.229919Z","shell.execute_reply":"2022-07-23T15:06:40.877230Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.fit(X_t, y_t)\nY_val_pred = model.predict(X_val)\n\nfrom sklearn.metrics import ConfusionMatrixDisplay\n\nConfusionMatrixDisplay.from_predictions(y_val, Y_val_pred)\nplt.title(\"Confusion matrix for training set\")\nplt.show()\n\nprint(accuracy_score(y_val, Y_val_pred))","metadata":{"execution":{"iopub.status.busy":"2022-07-23T15:06:40.880299Z","iopub.execute_input":"2022-07-23T15:06:40.880741Z","iopub.status.idle":"2022-07-23T15:06:41.234484Z","shell.execute_reply.started":"2022-07-23T15:06:40.880696Z","shell.execute_reply":"2022-07-23T15:06:41.233244Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"This algorithm has a better cross validation score, we'll keep it. The next step would be to optimize the hyper-parameters","metadata":{}},{"cell_type":"code","source":"make_submission()","metadata":{"execution":{"iopub.status.busy":"2022-07-23T15:06:41.236921Z","iopub.execute_input":"2022-07-23T15:06:41.237277Z","iopub.status.idle":"2022-07-23T15:06:41.282513Z","shell.execute_reply.started":"2022-07-23T15:06:41.237246Z","shell.execute_reply":"2022-07-23T15:06:41.281482Z"},"trusted":true},"execution_count":null,"outputs":[]}]}