{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"### References\nThis notebook was adapted from the following resources.\n- [Titanic Data Science Solutions](https://www.kaggle.com/code/startupsci/titanic-data-science-solutions) by Manav Sehgal\n- [Titanic Tutorial](https://www.kaggle.com/code/alexisbcook/titanic-tutorial)  by Alexis Cook\n- [IBM Data Science Course](https://www.coursera.org/professional-certificates/ibm-data-science) on Coursera","metadata":{"papermill":{"duration":0.016211,"end_time":"2022-07-29T12:56:30.500276","exception":false,"start_time":"2022-07-29T12:56:30.484065","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"# Part 0: Household stuff","metadata":{"execution":{"iopub.execute_input":"2022-07-27T07:41:52.154241Z","iopub.status.busy":"2022-07-27T07:41:52.153515Z","iopub.status.idle":"2022-07-27T07:41:52.158166Z","shell.execute_reply":"2022-07-27T07:41:52.157298Z","shell.execute_reply.started":"2022-07-27T07:41:52.154200Z"},"papermill":{"duration":0.014324,"end_time":"2022-07-29T12:56:30.529350","exception":false,"start_time":"2022-07-29T12:56:30.515026","status":"completed"},"tags":[]}},{"cell_type":"code","source":"# Kaggle.\nPATH = \"/kaggle/input/titanic/\"\n\n# JupyterLab - to format code in Black.\n# %load_ext lab_black\n# PATH = \"\"","metadata":{"papermill":{"duration":0.032079,"end_time":"2022-07-29T12:56:30.576560","exception":false,"start_time":"2022-07-29T12:56:30.544481","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-30T10:26:03.068638Z","iopub.execute_input":"2022-07-30T10:26:03.069058Z","iopub.status.idle":"2022-07-30T10:26:03.074081Z","shell.execute_reply.started":"2022-07-30T10:26:03.069016Z","shell.execute_reply":"2022-07-30T10:26:03.072676Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Part 1: Import libraries & load data","metadata":{"papermill":{"duration":0.014241,"end_time":"2022-07-29T12:56:30.605491","exception":false,"start_time":"2022-07-29T12:56:30.591250","status":"completed"},"tags":[]}},{"cell_type":"code","source":"# Data analysis and wrangling.\nimport numpy as np\nimport pandas as pd\nimport random as rnd\n\n# Visualisation.\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\n%matplotlib inline\n\n# Preprocess data.\nfrom sklearn import preprocessing\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.pipeline import make_pipeline\nfrom sklearn.preprocessing import StandardScaler\n\n# Machine learning.\nfrom sklearn.model_selection import GridSearchCV\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.svm import SVC, LinearSVC\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.neighbors import KNeighborsClassifier\nfrom sklearn.tree import DecisionTreeClassifier","metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","papermill":{"duration":1.651091,"end_time":"2022-07-29T12:56:32.271013","exception":false,"start_time":"2022-07-29T12:56:30.619922","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-30T10:26:03.222625Z","iopub.execute_input":"2022-07-30T10:26:03.223775Z","iopub.status.idle":"2022-07-30T10:26:03.238938Z","shell.execute_reply.started":"2022-07-30T10:26:03.223719Z","shell.execute_reply":"2022-07-30T10:26:03.237129Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load training data.\ntrain_data = pd.read_csv(PATH + \"train.csv\")\ntrain_data.head(2)","metadata":{"papermill":{"duration":0.059518,"end_time":"2022-07-29T12:56:32.345203","exception":false,"start_time":"2022-07-29T12:56:32.285685","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-30T10:26:03.299399Z","iopub.execute_input":"2022-07-30T10:26:03.300273Z","iopub.status.idle":"2022-07-30T10:26:03.323973Z","shell.execute_reply.started":"2022-07-30T10:26:03.300233Z","shell.execute_reply":"2022-07-30T10:26:03.322833Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load test data.\ntest_data = pd.read_csv(PATH + \"test.csv\")\ntest_data.head(2)","metadata":{"papermill":{"duration":0.043253,"end_time":"2022-07-29T12:56:32.403380","exception":false,"start_time":"2022-07-29T12:56:32.360127","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-30T10:26:03.377386Z","iopub.execute_input":"2022-07-30T10:26:03.378041Z","iopub.status.idle":"2022-07-30T10:26:03.398440Z","shell.execute_reply.started":"2022-07-30T10:26:03.377991Z","shell.execute_reply":"2022-07-30T10:26:03.397573Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Part 2: Explore data\n## 2.1 Get high-level sense of data","metadata":{"papermill":{"duration":0.014659,"end_time":"2022-07-29T12:56:32.433480","exception":false,"start_time":"2022-07-29T12:56:32.418821","status":"completed"},"tags":[]}},{"cell_type":"code","source":"# Get a sense of all data.\ntrain_data.info()\nprint(\"_\" * 40)\ntest_data.info()","metadata":{"papermill":{"duration":0.058197,"end_time":"2022-07-29T12:56:32.506789","exception":false,"start_time":"2022-07-29T12:56:32.448592","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-30T10:26:03.438675Z","iopub.execute_input":"2022-07-30T10:26:03.439360Z","iopub.status.idle":"2022-07-30T10:26:03.461356Z","shell.execute_reply.started":"2022-07-30T10:26:03.439324Z","shell.execute_reply":"2022-07-30T10:26:03.460469Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Observations:\n- Training set: Missing values in Age, Cabin and Embarked\n- Test set: Missing values in Age, Fare and Cabin","metadata":{"execution":{"iopub.execute_input":"2022-07-27T07:56:21.199026Z","iopub.status.busy":"2022-07-27T07:56:21.198597Z","iopub.status.idle":"2022-07-27T07:56:21.205059Z","shell.execute_reply":"2022-07-27T07:56:21.203976Z","shell.execute_reply.started":"2022-07-27T07:56:21.198993Z"},"papermill":{"duration":0.014832,"end_time":"2022-07-29T12:56:32.536998","exception":false,"start_time":"2022-07-29T12:56:32.522166","status":"completed"},"tags":[]}},{"cell_type":"code","source":"# Understand distribution of numerical training data.\ntrain_data.describe()","metadata":{"papermill":{"duration":0.05353,"end_time":"2022-07-29T12:56:32.605816","exception":false,"start_time":"2022-07-29T12:56:32.552286","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-30T10:26:03.519095Z","iopub.execute_input":"2022-07-30T10:26:03.519777Z","iopub.status.idle":"2022-07-30T10:26:03.553772Z","shell.execute_reply.started":"2022-07-30T10:26:03.519737Z","shell.execute_reply":"2022-07-30T10:26:03.552600Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Understand distribution of numerical test data.\ntest_data.describe()","metadata":{"papermill":{"duration":0.049874,"end_time":"2022-07-29T12:56:32.671353","exception":false,"start_time":"2022-07-29T12:56:32.621479","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-30T10:26:03.614881Z","iopub.execute_input":"2022-07-30T10:26:03.615694Z","iopub.status.idle":"2022-07-30T10:26:03.647457Z","shell.execute_reply.started":"2022-07-30T10:26:03.615655Z","shell.execute_reply":"2022-07-30T10:26:03.646569Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Understand distribution of categorical training data.\ntrain_data.describe(include=[\"O\"])","metadata":{"papermill":{"duration":0.042588,"end_time":"2022-07-29T12:56:32.729609","exception":false,"start_time":"2022-07-29T12:56:32.687021","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-30T10:26:03.758389Z","iopub.execute_input":"2022-07-30T10:26:03.759048Z","iopub.status.idle":"2022-07-30T10:26:03.782172Z","shell.execute_reply.started":"2022-07-30T10:26:03.759002Z","shell.execute_reply":"2022-07-30T10:26:03.781025Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Understand distribution of categorical test data.\ntest_data.describe(include=[\"O\"])","metadata":{"papermill":{"duration":0.040767,"end_time":"2022-07-29T12:56:32.786451","exception":false,"start_time":"2022-07-29T12:56:32.745684","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-30T10:26:03.892888Z","iopub.execute_input":"2022-07-30T10:26:03.893787Z","iopub.status.idle":"2022-07-30T10:26:03.916281Z","shell.execute_reply.started":"2022-07-30T10:26:03.893751Z","shell.execute_reply":"2022-07-30T10:26:03.915501Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 2.2 Deep dive to validate certain observations","metadata":{"execution":{"iopub.execute_input":"2022-07-27T08:14:01.414217Z","iopub.status.busy":"2022-07-27T08:14:01.413797Z","iopub.status.idle":"2022-07-27T08:14:01.419378Z","shell.execute_reply":"2022-07-27T08:14:01.418137Z","shell.execute_reply.started":"2022-07-27T08:14:01.414184Z"},"papermill":{"duration":0.015662,"end_time":"2022-07-29T12:56:32.818294","exception":false,"start_time":"2022-07-29T12:56:32.802632","status":"completed"},"tags":[]}},{"cell_type":"code","source":"# How does survival correlate with class?\ntrain_data[[\"Pclass\", \"Survived\"]].groupby(\n    [\"Pclass\"], as_index=False\n).mean().sort_values(by=\"Survived\", ascending=False)","metadata":{"papermill":{"duration":0.03534,"end_time":"2022-07-29T12:56:32.869611","exception":false,"start_time":"2022-07-29T12:56:32.834271","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-30T10:26:03.981044Z","iopub.execute_input":"2022-07-30T10:26:03.982079Z","iopub.status.idle":"2022-07-30T10:26:03.995636Z","shell.execute_reply.started":"2022-07-30T10:26:03.982037Z","shell.execute_reply":"2022-07-30T10:26:03.994347Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# How does survival correlate with sex?\ntrain_data[[\"Sex\", \"Survived\"]].groupby([\"Sex\"], as_index=False).mean().sort_values(\n    by=\"Survived\", ascending=False\n)","metadata":{"papermill":{"duration":0.034092,"end_time":"2022-07-29T12:56:32.920040","exception":false,"start_time":"2022-07-29T12:56:32.885948","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-30T10:26:04.078787Z","iopub.execute_input":"2022-07-30T10:26:04.079232Z","iopub.status.idle":"2022-07-30T10:26:04.095291Z","shell.execute_reply.started":"2022-07-30T10:26:04.079200Z","shell.execute_reply":"2022-07-30T10:26:04.094310Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# How does survival correlate with number of siblings?\ntrain_data[[\"SibSp\", \"Survived\"]].groupby([\"SibSp\"], as_index=False).mean().sort_values(\n    by=\"Survived\", ascending=False\n)","metadata":{"papermill":{"duration":0.036684,"end_time":"2022-07-29T12:56:32.973152","exception":false,"start_time":"2022-07-29T12:56:32.936468","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-30T10:26:04.173926Z","iopub.execute_input":"2022-07-30T10:26:04.174306Z","iopub.status.idle":"2022-07-30T10:26:04.190301Z","shell.execute_reply.started":"2022-07-30T10:26:04.174273Z","shell.execute_reply":"2022-07-30T10:26:04.188767Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# How does survival correlate with number of parents/children?\ntrain_data[[\"Parch\", \"Survived\"]].groupby([\"Parch\"], as_index=False).mean().sort_values(\n    by=\"Survived\", ascending=False\n)","metadata":{"papermill":{"duration":0.035055,"end_time":"2022-07-29T12:56:33.025005","exception":false,"start_time":"2022-07-29T12:56:32.989950","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-30T10:26:04.289322Z","iopub.execute_input":"2022-07-30T10:26:04.289719Z","iopub.status.idle":"2022-07-30T10:26:04.304519Z","shell.execute_reply.started":"2022-07-30T10:26:04.289686Z","shell.execute_reply":"2022-07-30T10:26:04.303548Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Observations:\n- Upper class passengers did have a higher chance of survival.\n- Females also had a higher chance of survival.\n- The above two make sense, but I don\"t get why there was a negative correlation between number of siblings and survival - was it because they were trying to find their siblings when the ship was sinking?","metadata":{"papermill":{"duration":0.016426,"end_time":"2022-07-29T12:56:33.058138","exception":false,"start_time":"2022-07-29T12:56:33.041712","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"## 2.3 Visualise the data","metadata":{"papermill":{"duration":0.016278,"end_time":"2022-07-29T12:56:33.091248","exception":false,"start_time":"2022-07-29T12:56:33.074970","status":"completed"},"tags":[]}},{"cell_type":"code","source":"# Correlate numerical features.\ngrid = sns.FacetGrid(train_data, col=\"Survived\")\ngrid.map(plt.hist, \"Age\", alpha=0.5, bins=20)","metadata":{"papermill":{"duration":0.495156,"end_time":"2022-07-29T12:56:33.603088","exception":false,"start_time":"2022-07-29T12:56:33.107932","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-30T10:26:04.380690Z","iopub.execute_input":"2022-07-30T10:26:04.381273Z","iopub.status.idle":"2022-07-30T10:26:04.762884Z","shell.execute_reply.started":"2022-07-30T10:26:04.381227Z","shell.execute_reply":"2022-07-30T10:26:04.761558Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Observations:\n- Infants had a high survival rate\n- Many 15-30 year olds did not survive \n- It\"s interesting that the age groups which are most societally productive/independent were deprioritised in the limited spots for rescue. I understand why, but can\"t help but highlight the irony.","metadata":{"papermill":{"duration":0.016901,"end_time":"2022-07-29T12:56:33.637286","exception":false,"start_time":"2022-07-29T12:56:33.620385","status":"completed"},"tags":[]}},{"cell_type":"code","source":"# Correlate numerical and ordinal features.\ngrid = sns.FacetGrid(train_data, col=\"Survived\", row=\"Pclass\")\ngrid.map(plt.hist, \"Age\", alpha=0.5, bins=20)\ngrid.add_legend()","metadata":{"papermill":{"duration":1.447997,"end_time":"2022-07-29T12:56:35.102350","exception":false,"start_time":"2022-07-29T12:56:33.654353","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-30T10:26:04.764733Z","iopub.execute_input":"2022-07-30T10:26:04.765050Z","iopub.status.idle":"2022-07-30T10:26:06.081936Z","shell.execute_reply.started":"2022-07-30T10:26:04.765023Z","shell.execute_reply":"2022-07-30T10:26:06.080851Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Observations:\n- Class 3 has most passengers, unfortunately, most did not survive (except infants).\n- Most class 1 passengers survived.","metadata":{"papermill":{"duration":0.017347,"end_time":"2022-07-29T12:56:35.137301","exception":false,"start_time":"2022-07-29T12:56:35.119954","status":"completed"},"tags":[]}},{"cell_type":"code","source":"# Correlate categorial features.\ngrid = sns.FacetGrid(train_data, row=\"Embarked\")\ngrid.map(\n    sns.pointplot,\n    \"Pclass\",\n    \"Survived\",\n    \"Sex\",\n    order=[1, 2, 3],\n    hue_order=[\"female\", \"male\"],\n)\ngrid.add_legend()","metadata":{"papermill":{"duration":1.290432,"end_time":"2022-07-29T12:56:36.445431","exception":false,"start_time":"2022-07-29T12:56:35.154999","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-30T10:26:06.083287Z","iopub.execute_input":"2022-07-30T10:26:06.084100Z","iopub.status.idle":"2022-07-30T10:26:07.130726Z","shell.execute_reply.started":"2022-07-30T10:26:06.084055Z","shell.execute_reply":"2022-07-30T10:26:07.129515Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Observations:\n- Female passengers had a much much better survival rate than men.\n","metadata":{"papermill":{"duration":0.017819,"end_time":"2022-07-29T12:56:36.481654","exception":false,"start_time":"2022-07-29T12:56:36.463835","status":"completed"},"tags":[]}},{"cell_type":"code","source":"# Correlate categorical and numerical features.\ngrid = sns.FacetGrid(train_data, row=\"Embarked\", col=\"Survived\")\ngrid.map(sns.barplot, \"Sex\", \"Fare\", order=[\"female\", \"male\"], alpha=0.5)\ngrid.add_legend()","metadata":{"papermill":{"duration":1.540498,"end_time":"2022-07-29T12:56:38.040216","exception":false,"start_time":"2022-07-29T12:56:36.499718","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-30T10:26:07.134020Z","iopub.execute_input":"2022-07-30T10:26:07.134682Z","iopub.status.idle":"2022-07-30T10:26:08.341893Z","shell.execute_reply.started":"2022-07-30T10:26:07.134636Z","shell.execute_reply":"2022-07-30T10:26:08.340661Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Observations:\n- Passengers who paid more for their tickets had higher chances of survival.\n- Port of embarkation affects survival rates.","metadata":{"papermill":{"duration":0.019113,"end_time":"2022-07-29T12:56:38.078227","exception":false,"start_time":"2022-07-29T12:56:38.059114","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"# Part 3: Wrangle data\n## 3.1 Complete features with missing/null values","metadata":{"papermill":{"duration":0.018418,"end_time":"2022-07-29T12:56:38.115905","exception":false,"start_time":"2022-07-29T12:56:38.097487","status":"completed"},"tags":[]}},{"cell_type":"code","source":"# Complete missing numerical values.\n# First understand the distribution for sex and age. Hopefully it\"s normal.\ngrid = sns.FacetGrid(train_data, row=\"Pclass\", col=\"Sex\")\ngrid.map(plt.hist, \"Age\", bins=20)\ngrid.add_legend()","metadata":{"papermill":{"duration":1.46843,"end_time":"2022-07-29T12:56:39.603039","exception":false,"start_time":"2022-07-29T12:56:38.134609","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-30T10:26:08.343589Z","iopub.execute_input":"2022-07-30T10:26:08.343935Z","iopub.status.idle":"2022-07-30T10:26:09.654178Z","shell.execute_reply.started":"2022-07-30T10:26:08.343894Z","shell.execute_reply":"2022-07-30T10:26:09.653119Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Fill out missing values for Age with the median age for that entry\"s Pclass & Sex.\nage_input = (\n    train_data[[\"Sex\", \"Pclass\", \"Age\"]]\n    .groupby(by=[\"Sex\", \"Pclass\"], dropna=True,)\n    .median()\n    .reset_index()\n)\n# Fill out missing values for training data.\ntrain_data = pd.merge(train_data, age_input, how=\"left\", on=[\"Sex\", \"Pclass\"])\ntrain_data[\"Age\"] = np.where(\n    train_data.Age_x.isnull(), train_data.Age_y, train_data.Age_x\n)\n# Do the same for test data.\ntest_data = pd.merge(test_data, age_input, how=\"left\", on=[\"Sex\", \"Pclass\"])\ntest_data[\"Age\"] = np.where(test_data.Age_x.isnull(), test_data.Age_y, test_data.Age_x)","metadata":{"papermill":{"duration":0.051853,"end_time":"2022-07-29T12:56:39.674694","exception":false,"start_time":"2022-07-29T12:56:39.622841","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-30T10:26:09.655756Z","iopub.execute_input":"2022-07-30T10:26:09.656075Z","iopub.status.idle":"2022-07-30T10:26:09.676823Z","shell.execute_reply.started":"2022-07-30T10:26:09.656044Z","shell.execute_reply":"2022-07-30T10:26:09.675728Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Fill out missing values for fare with the median fare for that entry\"s Pclass.\nfare_input = (\n    train_data[[\"Fare\", \"Pclass\"]]\n    .groupby(by=[\"Pclass\"], dropna=True,)\n    .median()\n    .reset_index()\n)\n# Fill out missing values for training data.\ntrain_data = pd.merge(train_data, fare_input, how=\"left\", on=[\"Pclass\"])\ntrain_data[\"Fare\"] = np.where(\n    train_data.Fare_x.isnull(), train_data.Fare_y, train_data.Fare_x\n)\n# Do the same for test data.\ntest_data = pd.merge(test_data, fare_input, how=\"left\", on=[\"Pclass\"])\ntest_data[\"Fare\"] = np.where(\n    test_data.Fare_x.isnull(), test_data.Fare_y, test_data.Fare_x\n)","metadata":{"papermill":{"duration":0.046591,"end_time":"2022-07-29T12:56:39.740593","exception":false,"start_time":"2022-07-29T12:56:39.694002","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-30T10:26:09.678395Z","iopub.execute_input":"2022-07-30T10:26:09.678734Z","iopub.status.idle":"2022-07-30T10:26:09.698449Z","shell.execute_reply.started":"2022-07-30T10:26:09.678705Z","shell.execute_reply":"2022-07-30T10:26:09.697341Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Complete missing categorical features.\nfor dataset in [train_data, test_data]:\n    dataset.Embarked = dataset.Embarked.fillna(train_data.Embarked.dropna().mode()[0])\ntrain_data[[\"Embarked\", \"Survived\"]].groupby(\n    [\"Embarked\"], as_index=False\n).mean().sort_values(by=\"Survived\", ascending=False)","metadata":{"papermill":{"duration":0.043969,"end_time":"2022-07-29T12:56:39.803988","exception":false,"start_time":"2022-07-29T12:56:39.760019","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-30T10:26:09.700139Z","iopub.execute_input":"2022-07-30T10:26:09.700455Z","iopub.status.idle":"2022-07-30T10:26:09.717419Z","shell.execute_reply.started":"2022-07-30T10:26:09.700426Z","shell.execute_reply":"2022-07-30T10:26:09.716693Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 3.2 Create new features from existing","metadata":{"papermill":{"duration":0.019931,"end_time":"2022-07-29T12:56:39.843954","exception":false,"start_time":"2022-07-29T12:56:39.824023","status":"completed"},"tags":[]}},{"cell_type":"code","source":"# Extract titles from names\nfor dataset in [train_data, test_data]:\n    dataset[\"Title\"] = dataset.Name.str.extract(\" ([A-Za-z]+)\\.\", expand=False)\npd.crosstab(train_data.Title, train_data.Sex, margins=True).sort_values(\n    \"All\", ascending=False\n)","metadata":{"papermill":{"duration":0.080207,"end_time":"2022-07-29T12:56:39.943650","exception":false,"start_time":"2022-07-29T12:56:39.863443","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-30T10:26:09.718457Z","iopub.execute_input":"2022-07-30T10:26:09.718781Z","iopub.status.idle":"2022-07-30T10:26:09.764409Z","shell.execute_reply.started":"2022-07-30T10:26:09.718754Z","shell.execute_reply":"2022-07-30T10:26:09.763199Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Make title more useful\nfor dataset in [train_data, test_data]:\n    dataset.Title = (\n        dataset.Title.replace(\n            [\n                \"Lady\",\n                \"Countess\",\n                \"Capt\",\n                \"Col\",\n                \"Don\",\n                \"Dr\",\n                \"Major\",\n                \"Rev\",\n                \"Sir\",\n                \"Jonkheer\",\n                \"Dona\",\n            ],\n            \"Rare\",\n        )\n        .replace(\"Mlle\", \"Miss\")\n        .replace(\"Ms\", \"Miss\")\n        .replace(\"Mme\", \"Mrs\")\n    )\ntrain_data[[\"Title\", \"Survived\"]].groupby([\"Title\"], as_index=False).mean().sort_values(\n    \"Survived\", ascending=False\n)","metadata":{"papermill":{"duration":0.050732,"end_time":"2022-07-29T12:56:40.014728","exception":false,"start_time":"2022-07-29T12:56:39.963996","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-30T10:26:09.767805Z","iopub.execute_input":"2022-07-30T10:26:09.768127Z","iopub.status.idle":"2022-07-30T10:26:09.790834Z","shell.execute_reply.started":"2022-07-30T10:26:09.768099Z","shell.execute_reply":"2022-07-30T10:26:09.789732Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Convert string to numerical representation to enable modelling.\n# This is technically a boolean field is_female, where 1 is True.\nfor dataset in [train_data, test_data]:\n    dataset['Sex'] = dataset['Sex'].map({'female': 1, 'male': 0}).astype(int)","metadata":{"execution":{"iopub.status.busy":"2022-07-30T10:26:09.792208Z","iopub.execute_input":"2022-07-30T10:26:09.792653Z","iopub.status.idle":"2022-07-30T10:26:09.802100Z","shell.execute_reply.started":"2022-07-30T10:26:09.792611Z","shell.execute_reply":"2022-07-30T10:26:09.801079Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Get dummies for categorical features\nfeatures_to_dummy = [\"Title\", \"Embarked\"]\n# Get dummies for training data.\none_hot = pd.get_dummies(train_data[features_to_dummy])\ntrain_data = train_data.join(one_hot)\ntrain_data.head()\n# Get dummies for test data.\none_hot = pd.get_dummies(test_data[features_to_dummy])\ntest_data = test_data.join(one_hot)","metadata":{"papermill":{"duration":0.044873,"end_time":"2022-07-29T12:56:40.079840","exception":false,"start_time":"2022-07-29T12:56:40.034967","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-30T10:26:09.805364Z","iopub.execute_input":"2022-07-30T10:26:09.806005Z","iopub.status.idle":"2022-07-30T10:26:09.824280Z","shell.execute_reply.started":"2022-07-30T10:26:09.805969Z","shell.execute_reply":"2022-07-30T10:26:09.823548Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create a new feature, \"FamilySize\"\nfor dataset in [train_data, test_data]:\n    dataset[\"FamilySize\"] = (\n        dataset[\"SibSp\"] + dataset[\"Parch\"] + 1\n    )  # Add 1 to count that individual.\ntrain_data[[\"FamilySize\", \"Survived\"]].groupby(\n    [\"FamilySize\"], as_index=False\n).mean().sort_values(by=\"Survived\", ascending=False)","metadata":{"papermill":{"duration":0.041707,"end_time":"2022-07-29T12:56:40.141954","exception":false,"start_time":"2022-07-29T12:56:40.100247","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-30T10:26:09.825260Z","iopub.execute_input":"2022-07-30T10:26:09.826183Z","iopub.status.idle":"2022-07-30T10:26:09.844269Z","shell.execute_reply.started":"2022-07-30T10:26:09.826153Z","shell.execute_reply":"2022-07-30T10:26:09.843284Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create a new feature, \"IsAlone\"\nfor dataset in [train_data, test_data]:\n    dataset[\"IsAlone\"] = 0\n    dataset.loc[dataset.FamilySize == 1, \"IsAlone\"] = 1\ntrain_data[[\"IsAlone\", \"Survived\"]].groupby(\n    [\"IsAlone\"], as_index=False\n).mean().sort_values(by=\"Survived\", ascending=False)","metadata":{"papermill":{"duration":0.042198,"end_time":"2022-07-29T12:56:40.204322","exception":false,"start_time":"2022-07-29T12:56:40.162124","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-30T10:26:09.845322Z","iopub.execute_input":"2022-07-30T10:26:09.846058Z","iopub.status.idle":"2022-07-30T10:26:09.862438Z","shell.execute_reply.started":"2022-07-30T10:26:09.846027Z","shell.execute_reply":"2022-07-30T10:26:09.861688Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 3.3 Correct data by dropping features","metadata":{"execution":{"iopub.execute_input":"2022-07-29T07:44:03.291410Z","iopub.status.busy":"2022-07-29T07:44:03.290251Z","iopub.status.idle":"2022-07-29T07:44:03.296460Z","shell.execute_reply":"2022-07-29T07:44:03.295466Z","shell.execute_reply.started":"2022-07-29T07:44:03.291352Z"},"papermill":{"duration":0.021438,"end_time":"2022-07-29T12:56:40.246588","exception":false,"start_time":"2022-07-29T12:56:40.225150","status":"completed"},"tags":[]}},{"cell_type":"code","source":"replaced_by_title = [\"Name\", \"PassengerId\"]\nreplaced_by_dummies = [\"Title\", \"Embarked\"]\nreplaced_by_is_alone = [\"Parch\", \"SibSp\"]\nreplaced_by_age = [\"Age_x\", \"Age_y\"]\nreplaced_by_fare = [\"Fare_x\", \"Fare_y\"]\nuninformative_cols = [\"Ticket\", \"Cabin\"]\n\ncolumns_to_drop = (\n    replaced_by_title\n    + replaced_by_dummies\n    + replaced_by_is_alone\n    + replaced_by_age\n    + replaced_by_fare\n    + uninformative_cols\n)\ntrain_data = train_data.drop(columns_to_drop, axis=1)\ntest_data_features = test_data.drop(columns_to_drop, axis=1)\ntrain_data.shape, test_data.shape","metadata":{"papermill":{"duration":0.041464,"end_time":"2022-07-29T12:56:40.310461","exception":false,"start_time":"2022-07-29T12:56:40.268997","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-30T10:26:09.863592Z","iopub.execute_input":"2022-07-30T10:26:09.864015Z","iopub.status.idle":"2022-07-30T10:26:09.875365Z","shell.execute_reply.started":"2022-07-30T10:26:09.863988Z","shell.execute_reply":"2022-07-30T10:26:09.874364Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Part 4: Model data and generate predictions\n","metadata":{"execution":{"iopub.execute_input":"2022-07-29T08:31:22.043206Z","iopub.status.busy":"2022-07-29T08:31:22.042705Z","iopub.status.idle":"2022-07-29T08:31:22.047784Z","shell.execute_reply":"2022-07-29T08:31:22.046477Z","shell.execute_reply.started":"2022-07-29T08:31:22.043172Z"},"papermill":{"duration":0.020421,"end_time":"2022-07-29T12:56:40.354894","exception":false,"start_time":"2022-07-29T12:56:40.334473","status":"completed"},"tags":[]}},{"cell_type":"code","source":"# Preprocess data. See https://scikit-learn.org/stable/modules/preprocessing.html\n\n# Split training data into train/test sets for model evaluation\nx_train, x_test, y_train, y_test = train_test_split(\n    train_data.drop(\"Survived\", axis=1), train_data[\"Survived\"], test_size=0.8\n)\n\n# Standardise dataset as models might behave badly if the individual features do not look like standard normally distributed data.\nscaler = preprocessing.StandardScaler().fit(x_train)\nx_train_scaled = scaler.transform(x_train)\nx_test_scaled = scaler.transform(x_test)","metadata":{"papermill":{"duration":0.038135,"end_time":"2022-07-29T12:56:40.413410","exception":false,"start_time":"2022-07-29T12:56:40.375275","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-30T10:26:09.876856Z","iopub.execute_input":"2022-07-30T10:26:09.877379Z","iopub.status.idle":"2022-07-30T10:26:09.893876Z","shell.execute_reply.started":"2022-07-30T10:26:09.877333Z","shell.execute_reply":"2022-07-30T10:26:09.892714Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 4.1 Logistic regression","metadata":{"papermill":{"duration":0.020292,"end_time":"2022-07-29T12:56:40.454653","exception":false,"start_time":"2022-07-29T12:56:40.434361","status":"completed"},"tags":[]}},{"cell_type":"code","source":"# Method 1: Raw log model.\nlog_reg = LogisticRegression()\nlog_reg.fit(x_train_scaled, y_train)\nacc_log_reg = log_reg.score(x_test_scaled, y_test) * 100\nacc_log_reg","metadata":{"papermill":{"duration":0.05072,"end_time":"2022-07-29T12:56:40.525693","exception":false,"start_time":"2022-07-29T12:56:40.474973","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-30T10:26:09.895679Z","iopub.execute_input":"2022-07-30T10:26:09.895991Z","iopub.status.idle":"2022-07-30T10:26:09.917241Z","shell.execute_reply.started":"2022-07-30T10:26:09.895962Z","shell.execute_reply":"2022-07-30T10:26:09.913273Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Validate our guesses of feature importance with the coefficients of the log model\ncoeff_df = pd.DataFrame(train_data.columns.delete(0))  # Delete \"Survived\"\ncoeff_df.columns = [\"Feature\"]\ncoeff_df[\"Correlation\"] = pd.Series(log_reg.coef_[0])\ncoeff_df.sort_values(by=\"Correlation\", ascending=False)","metadata":{"papermill":{"duration":0.064934,"end_time":"2022-07-29T12:56:40.629069","exception":false,"start_time":"2022-07-29T12:56:40.564135","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-30T10:26:09.919487Z","iopub.execute_input":"2022-07-30T10:26:09.919984Z","iopub.status.idle":"2022-07-30T10:26:09.941590Z","shell.execute_reply.started":"2022-07-30T10:26:09.919939Z","shell.execute_reply":"2022-07-30T10:26:09.940024Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Method 2: Alternative that uses Pipeline\npipe = Pipeline([(\"scaler\", StandardScaler()), (\"model\", LogisticRegression())])\npipe.fit(x_train, y_train)\nacc_log_reg = pipe.score(x_test, y_test) * 100\nacc_log_reg","metadata":{"papermill":{"duration":0.05107,"end_time":"2022-07-29T12:56:40.703293","exception":false,"start_time":"2022-07-29T12:56:40.652223","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-30T10:26:09.943746Z","iopub.execute_input":"2022-07-30T10:26:09.944256Z","iopub.status.idle":"2022-07-30T10:26:09.982889Z","shell.execute_reply.started":"2022-07-30T10:26:09.944213Z","shell.execute_reply":"2022-07-30T10:26:09.981418Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Validate our guesses of feature importance with the coefficients of the log model\nclassifier = pipe.named_steps[\"model\"]\ncoeff_df = pd.DataFrame(train_data.columns.delete(0))  # Delete \"Survived\"\ncoeff_df.columns = [\"Feature\"]\ncoeff_df[\"Correlation\"] = pd.Series(pd.Series(classifier.coef_[0]))\ncoeff_df.sort_values(by=\"Correlation\", ascending=False)","metadata":{"papermill":{"duration":0.072233,"end_time":"2022-07-29T12:56:40.815536","exception":false,"start_time":"2022-07-29T12:56:40.743303","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-30T10:26:09.984788Z","iopub.execute_input":"2022-07-30T10:26:09.985663Z","iopub.status.idle":"2022-07-30T10:26:10.007739Z","shell.execute_reply.started":"2022-07-30T10:26:09.985612Z","shell.execute_reply":"2022-07-30T10:26:10.006560Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Method 3: Alternative that uses Pipeline and GridSearch\nparameters = {\n    \"C\": [0.001, 0.01, 0.1, 1, 10],\n    \"penalty\": [\"l2\"],\n    \"solver\": [\"lbfgs\", \"newton-cg\", \"liblinear\"],\n}\nlogreg_classifier = make_pipeline(\n    StandardScaler(),\n    GridSearchCV(\n        LogisticRegression(), parameters, cv=5, refit=True\n    ),\n)\nlogreg_classifier.fit(x_train, y_train)\nlogreg_score = logreg_classifier.score(x_test, y_test) * 100\nlogreg_score","metadata":{"papermill":{"duration":0.05107,"end_time":"2022-07-29T12:56:40.703293","exception":false,"start_time":"2022-07-29T12:56:40.652223","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-30T10:26:10.009980Z","iopub.execute_input":"2022-07-30T10:26:10.010912Z","iopub.status.idle":"2022-07-30T10:26:10.397638Z","shell.execute_reply.started":"2022-07-30T10:26:10.010859Z","shell.execute_reply":"2022-07-30T10:26:10.396456Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 4.2 Support Vector Machines","metadata":{"execution":{"iopub.execute_input":"2022-07-29T12:02:04.974377Z","iopub.status.busy":"2022-07-29T12:02:04.973973Z","iopub.status.idle":"2022-07-29T12:02:04.978815Z","shell.execute_reply":"2022-07-29T12:02:04.977823Z","shell.execute_reply.started":"2022-07-29T12:02:04.974346Z"},"papermill":{"duration":0.020876,"end_time":"2022-07-29T12:56:40.857628","exception":false,"start_time":"2022-07-29T12:56:40.836752","status":"completed"},"tags":[]}},{"cell_type":"code","source":"parameters = {\n    \"kernel\": (\"linear\", \"rbf\", \"poly\", \"rbf\", \"sigmoid\"),\n    \"C\": np.logspace(-3, 3, 5),\n    \"gamma\": [\"scale\", \"auto\"],\n}\nsvm_classifier = make_pipeline(\n    StandardScaler(), GridSearchCV(SVC(), parameters, cv=5, refit=True),\n)\nsvm_classifier.fit(x_train, y_train)\nsvm_score = svm_classifier.score(x_test, y_test) * 100\nsvm_score","metadata":{"papermill":{"duration":0.047152,"end_time":"2022-07-29T12:56:40.925858","exception":false,"start_time":"2022-07-29T12:56:40.878706","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-30T10:26:10.399502Z","iopub.execute_input":"2022-07-30T10:26:10.400386Z","iopub.status.idle":"2022-07-30T10:26:14.895868Z","shell.execute_reply.started":"2022-07-30T10:26:10.400335Z","shell.execute_reply":"2022-07-30T10:26:14.894368Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 4.3 Random Forest","metadata":{"papermill":{"duration":0.020949,"end_time":"2022-07-29T12:56:40.968347","exception":false,"start_time":"2022-07-29T12:56:40.947398","status":"completed"},"tags":[]}},{"cell_type":"code","source":"parameters = {\n    \"n_estimators\": [300, 400],\n    \"criterion\": [\"gini\", \"entropy\"],\n    \"max_depth\": [5, 6, 7, 8, 9, 10, 11, 12],\n    \"max_features\": [\"log2\", \"sqrt\"],\n    \"min_samples_split\": [2, 5, 10],\n}\nrandom_forest_classifier = make_pipeline(  # RF performs better without scaling.\n    GridSearchCV(RandomForestClassifier(), parameters, cv=5, refit=True),\n)\nrandom_forest_classifier.fit(x_train, y_train)\nrandom_forest_score = random_forest_classifier.score(x_test, y_test) * 100\nrandom_forest_score","metadata":{"papermill":{"duration":0.224587,"end_time":"2022-07-29T12:56:41.214295","exception":false,"start_time":"2022-07-29T12:56:40.989708","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-30T10:42:26.870842Z","iopub.execute_input":"2022-07-30T10:42:26.871245Z","iopub.status.idle":"2022-07-30T10:46:43.054455Z","shell.execute_reply.started":"2022-07-30T10:42:26.871210Z","shell.execute_reply":"2022-07-30T10:46:43.053335Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 4.4 KNN","metadata":{"execution":{"iopub.execute_input":"2022-07-29T12:50:34.572425Z","iopub.status.busy":"2022-07-29T12:50:34.572001Z","iopub.status.idle":"2022-07-29T12:50:34.578070Z","shell.execute_reply":"2022-07-29T12:50:34.576894Z","shell.execute_reply.started":"2022-07-29T12:50:34.572381Z"},"papermill":{"duration":0.022095,"end_time":"2022-07-29T12:56:41.367871","exception":false,"start_time":"2022-07-29T12:56:41.345776","status":"completed"},"tags":[]}},{"cell_type":"code","source":"parameters = {\n    \"n_neighbors\": [5, 10, 15, 20, 25, 30],\n    \"algorithm\": [\"auto\", \"ball_tree\", \"kd_tree\", \"brute\"],\n    \"p\": [1, 2],\n    \"weights\": [\"uniform\", \"distance\"],\n}\nknn_classifier = make_pipeline(\n    StandardScaler(),\n    GridSearchCV(KNeighborsClassifier(), parameters, cv=5, refit=True),\n)\nknn_classifier.fit(x_train, y_train)\nknn_score = knn_classifier.score(x_test, y_test) * 100\nknn_score","metadata":{"papermill":{"duration":0.071455,"end_time":"2022-07-29T12:56:41.461257","exception":false,"start_time":"2022-07-29T12:56:41.389802","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-30T10:27:05.087796Z","iopub.execute_input":"2022-07-30T10:27:05.088521Z","iopub.status.idle":"2022-07-30T10:27:06.434727Z","shell.execute_reply.started":"2022-07-30T10:27:05.088455Z","shell.execute_reply":"2022-07-30T10:27:06.433542Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 4.5 Compare models","metadata":{"papermill":{"duration":0.030906,"end_time":"2022-07-29T12:56:41.516199","exception":false,"start_time":"2022-07-29T12:56:41.485293","status":"completed"},"tags":[]}},{"cell_type":"code","source":"model_scores_dict = {\n    \"model\": [\"knn\", \"random forest\", \"svm\", \"logistic reg\"],\n    \"score\": [knn_score, random_forest_score, svm_score, logreg_score],\n}\nmodel_scores = pd.DataFrame(model_scores_dict).sort_values(by=\"score\", ascending=False)\nmodel_scores\nsns.barplot(x=\"model\", y=\"score\", data=model_scores, color=\"purple\")","metadata":{"execution":{"iopub.status.busy":"2022-07-30T10:27:06.436619Z","iopub.execute_input":"2022-07-30T10:27:06.437380Z","iopub.status.idle":"2022-07-30T10:27:06.619048Z","shell.execute_reply.started":"2022-07-30T10:27:06.437337Z","shell.execute_reply":"2022-07-30T10:27:06.617688Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 4.6 Train all training data on the best model","metadata":{"papermill":{"duration":0.030906,"end_time":"2022-07-29T12:56:41.516199","exception":false,"start_time":"2022-07-29T12:56:41.485293","status":"completed"},"tags":[]}},{"cell_type":"code","source":"random_forest_classifier.fit(train_data.drop(\"Survived\", axis=1), train_data[\"Survived\"])\npredictions = random_forest_classifier.predict(test_data_features)","metadata":{"papermill":{"duration":0.24597,"end_time":"2022-07-29T12:56:41.784741","exception":false,"start_time":"2022-07-29T12:56:41.538771","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-30T10:27:06.620763Z","iopub.execute_input":"2022-07-30T10:27:06.621210Z","iopub.status.idle":"2022-07-30T10:28:07.137624Z","shell.execute_reply.started":"2022-07-30T10:27:06.621160Z","shell.execute_reply":"2022-07-30T10:28:07.136466Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Prepare output for competition submission","metadata":{"papermill":{"duration":0.021939,"end_time":"2022-07-29T12:56:41.828662","exception":false,"start_time":"2022-07-29T12:56:41.806723","status":"completed"},"tags":[]}},{"cell_type":"code","source":"output = pd.DataFrame({\"PassengerId\": test_data.PassengerId, \"Survived\": predictions})\noutput.to_csv(\"submission.csv\", index=False)\nprint(\"Your submission was successfully saved!\")","metadata":{"papermill":{"duration":0.037424,"end_time":"2022-07-29T12:56:41.887840","exception":false,"start_time":"2022-07-29T12:56:41.850416","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-30T10:28:07.138990Z","iopub.execute_input":"2022-07-30T10:28:07.139286Z","iopub.status.idle":"2022-07-30T10:28:07.148196Z","shell.execute_reply.started":"2022-07-30T10:28:07.139259Z","shell.execute_reply":"2022-07-30T10:28:07.147124Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}