{"cells":[{"metadata":{"_uuid":"0810d618d7c5c620e3348bd05a8a5f5528830a6f"},"cell_type":"markdown","source":"# Predicting Survival Rate from the Titanic Dataset - Preprocessing"},{"metadata":{"_uuid":"7ba35af4b20bec324e6b60dcab514558f3e4135c"},"cell_type":"markdown","source":"This kernel has been heavily influenced by other kernels, articles and discussions that I came across while figuring out how to proceed analysing this dataset."},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# Import the packages required to perform data analysis\n%matplotlib inline\nimport pandas as pd\nimport numpy as np\nimport seaborn as sns\nfrom matplotlib import pyplot as plt\n\n# Affects the appearence of matplotlib plots\nsns.set_palette(\"husl\")\nsns.set_style(\"whitegrid\")\n\npd.options.display.max_rows=None\npd.options.display.max_columns=None\n\nimport os\nprint(os.listdir(\"../input\"))\n\n# Any results you write to the current directory are saved as output.","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"# Import the train dataset\ndf_train = pd.read_csv('../input/train.csv')\n# Import the test dataset\ndf_test = pd.read_csv('../input/test.csv')\n# Save PassengerId for final submission\npassengerId = df_test.PassengerId\n# Join the train and test datasets for preprocessing\ndf_full= pd.concat([df_train, df_test], axis=0, ignore_index=True, sort=False)\n# Create indices to separate data later on\ntrain_idx = len(df_train)\ntest_idx = len(df_full) - len(df_test)\n# Make a copy of df_full\ndf = df_full.copy()\n# Eyeball the data\ndf.tail()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ce1c23b2276814387675af813c28171b6d7538a1"},"cell_type":"code","source":"# Study the features of the dataframe\ndf.info()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"5344bc646140669bdc1531c2a5af6a4026aa1ba1"},"cell_type":"code","source":"# Find missing values in each feature\ndf.isnull().sum()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"a22fba75aeaf8caa6b5a43df230826d59c24d5e4"},"cell_type":"markdown","source":"The features Age, Cabin and Embarked have missing values. Ignore the missing values in the feature Survived because it shows the missing values from the test dataset."},{"metadata":{"_uuid":"7eb5ea04af33661187976990675f855facabdfab"},"cell_type":"markdown","source":"## Feature Engineering"},{"metadata":{"trusted":true,"_uuid":"255ab618cb8e64910e7c28fc0df6b7b257723e6b"},"cell_type":"code","source":"# Get a list of the feature names\ndf.columns.values","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"a19ba14ad475f42c9278c739d68fc96385d92be7"},"cell_type":"markdown","source":"## Create a feature named 'Title' from the categorical feature 'Name'"},{"metadata":{"trusted":true,"_uuid":"4ca4a3796a855881b82d7d1b4ce5f0c290966f47"},"cell_type":"code","source":"# Building the logic to write the get_title function\nprint(df['Name'][0])\nprint(df['Name'][0].split(',')[1].split('.')[0].strip())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"292253e6354d0cc043f79809b369e7b22fdf552d"},"cell_type":"code","source":"# Function to extract titles from names\ndef get_title(name):\n    if '.' in name:\n        return name.split(',')[1].split('.')[0].strip()\n    else:\n        return 'unknown'","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"91f3f5d5237e1de617028f22288f8d3d9ef9a984"},"cell_type":"code","source":"# Applying the function get_title() on the feature 'Name' using a list comprehension\ntitles = pd.Series([x for x in df.Name.apply(lambda x : get_title(x))])\ntitle_count = titles.value_counts().sort_index()\ntitle_count","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"9720a5bd310afda0337a242bb977efe4bb3c40ec"},"cell_type":"code","source":"# Group the titles that have a low frequency of occurence into an array\nrare_titles = title_count[title_count <= 10]\nrare_titles.index.values","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"3c7610175b43f95c221f388227cae74bd82b6e09"},"cell_type":"code","source":"# Next, group the titles to be engineered\nrare_titles_Royalty = ['Lady', 'the Countess', 'Don', 'Dona', 'Jonkheer', 'Sir']\nrare_titles_Officer = ['Capt', 'Col', 'Major']\nrare_titles_Mr = ['Rev']\nrare_titles_Mrs = ['Mme']\nrare_titles_Miss = ['Mlle', 'Ms']\nrare_titles_Doctor = ['Dr']\n\n# Create a function to Normalize the titles\ndef normalize_titles(df):\n    title = df['Title']\n    \n    if title in rare_titles_Royalty:\n        return 'Royalty'\n    if title in rare_titles_Officer:\n        return 'Officer'\n    if title in rare_titles_Mr:\n        return 'Mr'\n    elif title in rare_titles_Mrs:\n        return 'Mrs'\n    elif title in rare_titles_Miss:\n        return 'Miss'\n    elif title in rare_titles_Doctor:\n        if df['Sex'] == 'male':\n            return 'Mr'\n        else: \n            return 'Mrs'\n    else: \n        return title","execution_count":null,"outputs":[]},{"metadata":{"_kg_hide-output":true,"trusted":true,"_uuid":"4bcb62ee8362f6d196f461c022a3adf2674eeb8e"},"cell_type":"code","source":"# Create a feature named Title in the train dataframe\ndf['Title'] = titles\ndf['Title'].describe()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ecbb8b8dce5a763e816d77d91ef3f90b8edc385a"},"cell_type":"code","source":"# Apply the engineered titles to the Title column\ndf['Title'] = df.apply(normalize_titles, axis = 1)\ndf['Title'].describe()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a6c75ca571eeac633074cedf57cce75e027fc475"},"cell_type":"code","source":"# Eyeball the dataframe\ndf.head()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"217bfed80880be473d05ef45bf13b7d022dde970"},"cell_type":"markdown","source":"### Next, we analyse the feature 'Age'"},{"metadata":{"trusted":true,"_uuid":"fd1072da1964765f335dc9f147581d52a2ba9270"},"cell_type":"code","source":"# Analysing the Age column\nprint(df['Age'].describe(), \"\\n\")\nprint(\"Number NaN values:\", df['Age'].isnull().sum())","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"ced1d379a6a17444eaf832abf904773b38efeb0c"},"cell_type":"markdown","source":"The 'Age' feature has 263 missing values. These values have to be imputed before using this feature in any analyses. The best way to imputed these missing values is to do the following:"},{"metadata":{"trusted":true,"_uuid":"a6ad0a6536edec646c1dc056a5a391093fd86b25"},"cell_type":"code","source":"# Group by Sex, Pclass, and Title \ngrouped = df.groupby(['Sex', 'Pclass', 'Title'])\ngrouped.Age.median()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"e32479fb62dab19e5ca06e70128c84b1b10ecfdb"},"cell_type":"markdown","source":"By grouping 'Age' with ['Sex', 'Pclass', 'Title'] and then taking the median value, we are able to get values that better reflect the original values missing rather than simply taking the median of the of the feature 'Age'. Then use the transform with groupby to apply the median values to the feature 'Age' as shown below."},{"metadata":{"trusted":true,"_uuid":"94cd0b58dd41424c8bb26ceca3bd84a7fef94227"},"cell_type":"code","source":"# Apply the grouped median value on the Age NaN\ndf.Age = grouped.Age.transform(lambda x: x.fillna(x.median(), inplace=False))","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"e41c30c40f2bfd09c4a344d9e899ed1e0369d12a"},"cell_type":"markdown","source":"### Moving on, we analyse the feature 'Cabin'"},{"metadata":{"trusted":true,"_uuid":"79c153fa1f973e9980dbbf5c7b6feab4d589866b"},"cell_type":"code","source":"# Analysing the feature named 'Cabin'\nprint(df.Cabin.describe(), \"\\n\")\nprint('Number of NaN values:', df.Cabin.isnull().sum())","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"20dd9ce8e57c4cfbf57178d3c71a5beddca45c87"},"cell_type":"markdown","source":"Cabin is a categorical feature and the total number of missing values in this feature is too high to be imputed successfully because the dataset does not provide sufficient information to impute this data. Therefore, we shall drop this column from our final preprocessed dataset."},{"metadata":{"_uuid":"8f2586641291771ea2de616e4dfe1651cb2cbd25"},"cell_type":"markdown","source":"### Analysing features 'Embarked' and 'Fare'"},{"metadata":{"trusted":true,"_uuid":"84a5d10cfe1e980175f76f8ccbb31c9967a4d445"},"cell_type":"code","source":"# Drop feature Cabin since it has few values and is unlikely to impact our analysis\ndf.drop(['Cabin'], axis=1, inplace=True)\n# Fill Embarked NaN values with the most frequently embarked location\ndf.Embarked = df.Embarked.fillna(df.Embarked.mode().values[0])\n# Fill Fare NaN values with median fare\ndf.Fare = round(df.Fare.fillna(df.Fare.median()))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f7f218ce749b0efa62b66119229365ff73f6fd3e"},"cell_type":"code","source":"# View changes\ndf.info()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"db9c07bdd646fbda4204cacb886b953374574af6"},"cell_type":"markdown","source":"### Create feature 'FamilySize'"},{"metadata":{"trusted":true,"_uuid":"bb3c14a8917d1170f1552961f729dcfbdb516520"},"cell_type":"code","source":"# Create feature FamilySize\ndf['FamilySize'] = df['SibSp'] + df['Parch'] + 1","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"f9ce96956c8105512b7ea2b5f237652d77cfd15e"},"cell_type":"markdown","source":"### Create a checkpoint-1"},{"metadata":{"trusted":true,"_uuid":"f20d5f5e99fc1a27875cc05f88b155f8c5eed27d"},"cell_type":"code","source":"# Checkpoint-1\ndf_before_dummies = df.copy()\n# Eyeball the dataframe\ndf.head()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"241eda265c0d5f3cde8d1ea8e02633352367f036"},"cell_type":"markdown","source":"## Exploratory Data Analysis"},{"metadata":{"_uuid":"3828e1a08269a519fa5cc8c1a842245fd32f892b"},"cell_type":"markdown","source":"In this section, we take a look at the relationship between the features in our dataset and their impact on survival."},{"metadata":{"_uuid":"34c6cb003db5cd62773408c68653e7e99a1739dc"},"cell_type":"markdown","source":"**1. Looking at Sex, Age and Survival**"},{"metadata":{"trusted":true,"_uuid":"657a8b8b6eb187dd8eb47a76fca1c415239ce2a3"},"cell_type":"code","source":"fig, axes = plt.subplots(nrows=1, ncols=2, figsize=(10,4))\n\nsurvived = 'Survived'\ndied = 'Died'\n\nwomen = df[df.Sex=='female']\nmen = df[df.Sex=='male']\n\nax = sns.distplot(women[women.Survived==1].Age.dropna(), ax=axes[0], bins = 20, label=survived, kde=False, color='c')\nax = sns.distplot(women[women.Survived==0].Age.dropna(), ax=axes[0], bins = 40, label=died, kde=False, color='r')\nax.legend()\nax.set_title('Female')\n\nax = sns.distplot(men[men.Survived==1].Age.dropna(), ax=axes[1], bins = 20, label=survived, kde=False, color='b')\nax = sns.distplot(men[men.Survived==0].Age.dropna(), ax=axes[1], bins = 40, label=died, kde=False, color='r')\nax.legend()\nax.set_title('Male')\n","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"428660259ee6d74443a47b79c6f7b9a91901f91a"},"cell_type":"markdown","source":"The above histogram shows that a higher number of women survived when compared to men. Women between 15 - 35 and men between 25 - 35 have a higher chance of survival."},{"metadata":{"_uuid":"75a64e88bee737e6ad3102ffc6f5593e95cb6180"},"cell_type":"markdown","source":"**2. Analysing Pclass and Survival**"},{"metadata":{"trusted":true,"_uuid":"d19ad86bc5b4f0283025d665cf7fd73898b17521","scrolled":true},"cell_type":"code","source":"# Create a catplot\nax = sns.catplot(x='Pclass', hue='Survived', col='Title', data=df, palette='inferno',  \n                 col_wrap=3, height=4, aspect=1, kind='count')","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"1b4edcb8c35c6c787c4bef9f46a6f54f5487974b"},"cell_type":"markdown","source":"The plot above shows that Pclass 1 and Pclass 2 passengers are more likely to survive than Pclass 3 passengers. Also, women survived more than men."},{"metadata":{"_uuid":"d1a7cd45b34e93a97e335690925ed149dc389cf1"},"cell_type":"markdown","source":"### Convert the feature 'Sex' from categorical data to numeric data"},{"metadata":{"trusted":true,"_uuid":"e940cfe442879f48309f20041e96d525e5ded520"},"cell_type":"code","source":"df.Sex = df.Sex.map({'male':0, 'female':1})","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"57d43ac673daf91827ebc3c001bdaee79f7de138"},"cell_type":"markdown","source":"### pd .get_dummies()"},{"metadata":{"trusted":true,"_uuid":"d3d1e76da45d0f952e509dfb1e46737b8959b19b"},"cell_type":"code","source":"# Create dummy variables for categorical features\nPclass_dummies = pd.get_dummies(df.Pclass, prefix='Pclass')\nTitle_dummies = pd.get_dummies(df.Title, prefix='Title')\nEmbarked_dummies = pd.get_dummies(df.Embarked, prefix='Embarked')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"acc3e754cb5eb523cc155fcae436b39cc277531a"},"cell_type":"code","source":"# Concatenate dummy columns with main dataset\ndf_dummies = pd.concat([df, Pclass_dummies, Title_dummies, Embarked_dummies], axis=1)\n# Drop categorical features\ndf_dummies.drop(columns={'PassengerId', 'Name', 'Ticket', 'Title', 'Embarked', 'Pclass', 'SibSp', 'Parch'}, inplace=True)\ndf_dummies.columns.values","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"35d54a8c1e44cf2c23c85233194e50d1fe8df29c"},"cell_type":"markdown","source":"## Create a checkpoint-2"},{"metadata":{"trusted":true,"_uuid":"ad8e581b77b41c84a192f6b5fd246d87ebbde404"},"cell_type":"code","source":"# Re-order the columns in the dataframe\ncols_reordered = ['Fare', 'Sex', 'Age', 'FamilySize', \n                  'Pclass_1', 'Pclass_2', 'Pclass_3', 'Title_Master',\n                  'Title_Miss', 'Title_Mr', 'Title_Mrs', 'Title_Officer',\n                  'Title_Royalty', 'Embarked_C', 'Embarked_Q', 'Embarked_S',\n                  'Survived']\n\n# Re-ordered columns and checkpoint-2\ndf_preprocessed = df_dummies[cols_reordered]\n\n# Convert features 'Fare' and 'Age' to datatype int\ndf_preprocessed.Age = df_preprocessed.Age.astype(int)\ndf_preprocessed.Fare = df_preprocessed.Fare.astype(int)\ndf_preprocessed.tail()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"a18487b0ce1775611c672e8435956d7ec7618599"},"cell_type":"markdown","source":"#### df_preprocessed is a cleaned dataframe which can be used to fit  machine learning models to predict the survival of passengers."},{"metadata":{"_uuid":"13e0e8a4b05c3592f75a00a8440f3dcfc1bdd2e0"},"cell_type":"markdown","source":"# Predicting Survival Rate from the Titanic Dataset - Machine Learning"},{"metadata":{"trusted":true,"_uuid":"8a45026bebe3a1e467d3e841b6c236ac9f7fc71c"},"cell_type":"code","source":"# Import the packages required for machine learning\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.model_selection import cross_val_score\nfrom sklearn import metrics\nfrom xgboost import XGBClassifier\nfrom xgboost import plot_importance\nfrom sklearn.feature_selection import SelectFromModel\nfrom sklearn.metrics import accuracy_score","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"33475456a4a0199489ba20f1a6cab4c31dc57da9"},"cell_type":"code","source":"# Create the train dataset\ntrain_df = df_preprocessed[:train_idx].copy()\nprint(\"Number of records:\",len(train_df))\n# Convert the feature Survived to int\ntrain_df.Survived = train_df.Survived.astype(int)\ntrain_df.tail()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"48bfd6c0927edf1646a15bbf7d7da45b6c954151"},"cell_type":"code","source":"# Create the test dataset\ntest_df = df_preprocessed[test_idx:].copy()\ntest_df.drop(['Survived'], axis=1, inplace=True)\nprint(\"Number of records:\",len(test_df))\ntest_df.tail()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"cb96420629ec95687617b60b2c293318463fa07c"},"cell_type":"markdown","source":"## Create the inputs and targets"},{"metadata":{"trusted":true,"_uuid":"65a573672c4699aa55b80b5d5527b229e765c2d0"},"cell_type":"code","source":"# Create the targets\ntargets = train_df.Survived.values\nprint(\"Length of the target array:\",len(targets))\nprint(targets.shape)\n\n# Create the inputs\ninputs = train_df.iloc[:, :-1]\nprint(\"Length of the input dataframe:\",len(inputs))\nprint(inputs.shape)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"163d43bb33f8085412f2c09ab8b6d921795abff7"},"cell_type":"markdown","source":"## Split the data"},{"metadata":{"trusted":true,"_uuid":"751c461a9a79057e552fe59d650cf205b6306c5b"},"cell_type":"code","source":"X_train, X_test, y_train, y_test = train_test_split(inputs, targets, test_size=418, shuffle=True, \n                                                   random_state=20, stratify=targets)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"c7ccd102330795704eafe6e64c53629fc79dbcc1"},"cell_type":"markdown","source":"## Model - XGBoost"},{"metadata":{"trusted":true,"_uuid":"285f1f0479e608ded8151bc6032c6b264de5052b"},"cell_type":"code","source":"# Instantiate the model\nxgb = XGBClassifier()\n# Fit the data\nxgb.fit(X_train, y_train)\n# Predict the values\nxgb_predictions = xgb.predict(X_train)\n# Display the accuracy score of the train dataset\nprint(\"Accuracy score: %.2f%%\" % (round(xgb.score(X_train, y_train) * 100)))","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"f6d326ba2594b00c730713f74f0050a6c5c1a3c0"},"cell_type":"markdown","source":"Next, check the accuracy score of the training model using cross-validation."},{"metadata":{"trusted":true,"_uuid":"72d958ca70f97e156126ac5f3cd4062368354cfc"},"cell_type":"code","source":"kfold = StratifiedKFold(n_splits=10, random_state=20, shuffle=True)\nresults = cross_val_score(xgb, X_train, y_train, cv=kfold)\nprint('Accuracy score \\nmean: %.2f%% \\nsd: %.2f%%' % (results.mean()*100, results.std()*100))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e3a264e5010ba4a10fc175d65ca3f959967a323f"},"cell_type":"code","source":"# Plot feature importance\nfig, axes = plt.subplots(figsize=(10, 8))\nplot_importance(xgb, ax=axes)\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"5ed55730e902f73559f75e06a27e56c293a109c7"},"cell_type":"markdown","source":"## Feature Selection with XGBoost Feature Importance Scores"},{"metadata":{"trusted":true,"_uuid":"61168acaf43e130def46a42a18efdbc6e7ba7bbe"},"cell_type":"code","source":"# fit model on all training data\nmodel = XGBClassifier()\nmodel.fit(X_train, y_train)\n\n# make predictions for test data and evaluate\ny_pred = model.predict(X_test)\npredictions = [round(value) for value in y_pred]\naccuracy = accuracy_score(y_test, predictions)\nprint(\"Accuracy: %.2f%%\" % (accuracy * 100.0))\n\n# Fit model using each importance as a threshold\nthresholds = sorted(model.feature_importances_)\nfor thresh in thresholds:\n    # select features using threshold\n    selection = SelectFromModel(model, threshold=thresh, prefit=True)\n    select_X_train = selection.transform(X_train)\n    # train model\n    selection_model = XGBClassifier()\n    selection_model.fit(select_X_train, y_train)\n    # eval model\n    select_X_test = selection.transform(X_test)\n    y_pred = selection_model.predict(select_X_test)\n    predictions = [round(value) for value in y_pred]\n    accuracy = accuracy_score(y_test, predictions)\n    print(\"Thresh=%.3f, n=%d, Accuracy: %.2f%%\" % (thresh, select_X_train.shape[1], accuracy*100.0))","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"42f4713050ce09c4a4c2f2f765995b8654e47ee4"},"cell_type":"markdown","source":"## Implementing the model on test data"},{"metadata":{"trusted":true,"_uuid":"b81c2c98bafa2612313f26e812f6340e8deb51f4"},"cell_type":"code","source":"# instantiate the model\nmodel = XGBClassifier()\nmodel.fit(X_train, y_train)\n# select features using threshold\nselection = SelectFromModel(model, threshold=0.002, prefit=True)\nselection_model = XGBClassifier()\nselect_X_train = selection.transform(X_train)\n# train model\nselection_model.fit(select_X_train, y_train)\n# test model\nselect_X_test = selection.transform(test_df)\ny_pred = selection_model.predict(select_X_test)\npredictions = [round(value) for value in y_pred]\naccuracy = accuracy_score(y_test, predictions)\nprint(\"Thresh=%.3f, n=%d, Accuracy: %.2f%%\" % (thresh, select_X_test.shape[1], accuracy*100.0))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f09429a91e151370fabb9ef4ed4fac1cd31114e1"},"cell_type":"code","source":"test_df['PassengerId'] = passengerId\ntest_df.reset_index(inplace=True)\ntest_df.drop(['index'], axis=1, inplace=True)\nsubmission_df = pd.concat([test_df.PassengerId, pd.DataFrame(y_pred)], axis=1)\nsubmission_df.columns = ['PassengerId', 'Survived']\nsubmission_df['PassengerId'] = passengerId\nsubmission_df.tail()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"72d04ea0fbfc6b9d338a9d7b0c6abe5c1b471b80"},"cell_type":"code","source":"submission_df.to_csv(\"../working/submission_titanic.csv\", index = False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"7aebdb9e3e5b69e448973bb72c96fd543b9c98f3"},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}