{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport random\nimport warnings\n\n\n#Statistical testing\nfrom scipy.stats import ttest_ind\nfrom scipy.stats import skew\n\n#Preprocessing\nfrom sklearn.preprocessing import OneHotEncoder\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.experimental import enable_iterative_imputer\nfrom sklearn.impute import IterativeImputer\nfrom sklearn import preprocessing\nfrom sklearn.model_selection import train_test_split\n\n#Feature engineering\nfrom nltk.corpus import names\nimport nltk\n\n#For model training\nfrom sklearn.model_selection import cross_val_score\nfrom sklearn.model_selection import GridSearchCV\nimport xgboost as xgb\nfrom xgboost import XGBClassifier\nfrom xgboost import plot_importance\n\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session\nwarnings.filterwarnings(\"ignore\")","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-25T13:23:19.954099Z","iopub.execute_input":"2022-07-25T13:23:19.954491Z","iopub.status.idle":"2022-07-25T13:23:19.967783Z","shell.execute_reply.started":"2022-07-25T13:23:19.954458Z","shell.execute_reply":"2022-07-25T13:23:19.966616Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Notebook Approach\n\nIn this notebook we explore a binary classification problem predicting whether passengers on a fictional starship were successfully transported to their final destination or if they were victims of a mid trip accident.\n\nThe notebook is structured as follows:\n\n1. Data Loading and initial inspection\n\n2. Data Cleaning and Transformation\n\n    - Imputing missing numeric values\n    - Normalizing numeric Values\n    - Imputing Missing Categorical values\n3. Feature Engineering\n\n    - Cabin and Passenger ID\n    - Extracting Gender from Name\n    - Bucketting Age\n4. Test/Train Split\n\n5. Hyperparameter tuning with XGBoost Classifier\n\n6. Final Model training\n\n7. Prediction and Submission of results","metadata":{}},{"cell_type":"markdown","source":"## Data Loading and Initial Inspection","metadata":{}},{"cell_type":"code","source":"df = pd.read_csv('../input/spaceship-titanic/train.csv')\ntest_df = pd.read_csv('../input/spaceship-titanic/test.csv')\n#Getting initial summary statistics\ndf.head(n=5)","metadata":{"execution":{"iopub.status.busy":"2022-07-25T13:23:20.019208Z","iopub.execute_input":"2022-07-25T13:23:20.020357Z","iopub.status.idle":"2022-07-25T13:23:20.097580Z","shell.execute_reply.started":"2022-07-25T13:23:20.020290Z","shell.execute_reply":"2022-07-25T13:23:20.096287Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Counting null values\nprint(\" \\nCount total NaN at each column in a DataFrame : \\n\\n\",\n      df.isnull().sum())\n#print(df.skew(numeric_only=True))","metadata":{"execution":{"iopub.status.busy":"2022-07-25T13:23:20.100027Z","iopub.execute_input":"2022-07-25T13:23:20.100472Z","iopub.status.idle":"2022-07-25T13:23:20.120591Z","shell.execute_reply.started":"2022-07-25T13:23:20.100427Z","shell.execute_reply":"2022-07-25T13:23:20.119221Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We can see that 14 of our input variables have missing values, prior to imputing these missing values we're going to add a binary column encoding if a given passenger had missing data:","metadata":{}},{"cell_type":"code","source":"df['had_missing'] = df.isnull().sum(axis =1)\ntest_df['had_missing'] = test_df.isnull().sum(axis =1)\n\n","metadata":{"execution":{"iopub.status.busy":"2022-07-25T13:23:20.122422Z","iopub.execute_input":"2022-07-25T13:23:20.123029Z","iopub.status.idle":"2022-07-25T13:23:20.143202Z","shell.execute_reply.started":"2022-07-25T13:23:20.122983Z","shell.execute_reply":"2022-07-25T13:23:20.142007Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Now visualizing distribution of numeric variables versus transportation\nnumerics = ['int16', 'int32', 'int64', 'float16', 'float32', 'float64']\nfor_visual = df.select_dtypes(include=numerics).copy()\nfor_visual['Transported'] = df['Transported']\ng = sns.pairplot(for_visual[['Age', 'RoomService', 'FoodCourt', 'ShoppingMall', 'Spa', 'VRDeck', 'Transported']],hue = 'Transported', corner = True)\ng.fig.suptitle(\"Correlation Between Numeric Variables and Transportation\", y = 1.01)\n\n","metadata":{"execution":{"iopub.status.busy":"2022-07-25T13:23:20.145618Z","iopub.execute_input":"2022-07-25T13:23:20.146354Z","iopub.status.idle":"2022-07-25T13:23:23.534842Z","shell.execute_reply.started":"2022-07-25T13:23:20.146309Z","shell.execute_reply":"2022-07-25T13:23:23.530664Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Examining the above pairplot we can see that there is a clear difference in Age between those transported and not - younger people are much more likely to be succesfully transported. For Roomservice, Spa, and VR deck we can see passengers with values of zero are disproportionally likely to not be transported, this likely indicates that more wealthy passengers have a higher probability of succesfull transportation.","metadata":{}},{"cell_type":"markdown","source":"## Data Cleaning P1 - Imputing Missing Numeric Values\n\nRather than imputing missing values based on column mean, median, or mode, we'll take a slighly more sophisticated approach and train a K Nearest neighbors model to do this imputation for us.\n\nWith this approach each missing value is calculated as the mean of the K neigbors that are closest to it in other dimensions.","metadata":{}},{"cell_type":"code","source":"#Imputing missing values using KNN Imputation\nnumeric_df = df.select_dtypes(include=numerics)\nmulti_imp = IterativeImputer(max_iter=9, random_state=42, verbose = 0,\n                            skip_complete = True, n_nearest_features = 10,\n                            tol = 0.001)\nmulti_imp.fit(numeric_df)\nnumeric_df[:] = multi_imp.transform(numeric_df)\n\n#Applying the same transform on our test set\nnumeric_test = test_df.select_dtypes(include=numerics)\nnumeric_test[:] = multi_imp.transform(numeric_test)","metadata":{"execution":{"iopub.status.busy":"2022-07-25T13:23:23.535989Z","iopub.status.idle":"2022-07-25T13:23:23.536589Z","shell.execute_reply.started":"2022-07-25T13:23:23.536279Z","shell.execute_reply":"2022-07-25T13:23:23.536304Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Data Cleaning P2 - Normalization\n\nHere we'll normalize our dataset and examine the skewness pre and post normalization","metadata":{}},{"cell_type":"code","source":"\nnormalized_numeric =  preprocessing.normalize(numeric_df)\nnormalized_numeric_df = pd.DataFrame(normalized_numeric , columns=numeric_df.columns)\n\n#Test set\nnormalized_numeric_test = preprocessing.normalize(numeric_test)\nnormalized_numeric_test_df = pd.DataFrame(normalized_numeric_test, columns=numeric_test.columns)","metadata":{"execution":{"iopub.status.busy":"2022-07-25T13:23:23.537924Z","iopub.status.idle":"2022-07-25T13:23:23.538454Z","shell.execute_reply.started":"2022-07-25T13:23:23.538186Z","shell.execute_reply":"2022-07-25T13:23:23.538211Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Measuring skewness of numeric variables pre and post transform","metadata":{}},{"cell_type":"code","source":"print(numeric_df.skew(axis = 0, skipna = True))\nprint(normalized_numeric_df.skew(axis = 0, skipna = True))","metadata":{"execution":{"iopub.status.busy":"2022-07-25T13:23:23.540405Z","iopub.status.idle":"2022-07-25T13:23:23.540968Z","shell.execute_reply.started":"2022-07-25T13:23:23.540662Z","shell.execute_reply":"2022-07-25T13:23:23.540688Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Data Cleaning P3 - Imputing Missing Categorical Variables\n\nFor the categorical data we'll use a simple mode imputer and fill nan's with the most common value in that column.","metadata":{}},{"cell_type":"code","source":"#performing imputation on train set\ncat_df = df.select_dtypes(include=object)\nmode_imp = SimpleImputer(missing_values=np.nan, strategy=\"most_frequent\")\nmode_imp.fit(cat_df)\ncat_df = pd.DataFrame(mode_imp.transform(cat_df), columns = cat_df.columns)\ncat_df.head(n=10)\n\n#performing imputation on test set\ntest_cat_df = test_df.select_dtypes(include=object)\ntest_cat_df = pd.DataFrame(mode_imp.transform(test_cat_df), columns = test_cat_df.columns)","metadata":{"execution":{"iopub.status.busy":"2022-07-25T13:23:23.542465Z","iopub.status.idle":"2022-07-25T13:23:23.543161Z","shell.execute_reply.started":"2022-07-25T13:23:23.542923Z","shell.execute_reply":"2022-07-25T13:23:23.542948Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Feature Engineering P1 - Cabin and Passenger\n\nExamining the categorical dataset we can see that some of them have many levels, cabin for example appears to have few repeated values which will make it difficult to use for prediction.\n\nTo address this we we'll apply feature engineering to split both the cabin and passenger ID into multiple columns such that Passenger 0 would have an ID of 0001 and then 01 whith a cabin of B, 0, P.","metadata":{}},{"cell_type":"code","source":"#Splitting train columns\ncat_df[['cabin_1', 'cabin_2', 'cabin_3']] = cat_df['Cabin'].str.split('/', expand=True)\ncat_df[['id_1', 'id_2']] = cat_df['PassengerId'].str.split('_', expand=True)\ncat_df = cat_df.drop(['Cabin', 'PassengerId'], axis = 1)\ncat_df.head()\n\n#Splitting test columns\ntest_cat_df[['cabin_1', 'cabin_2', 'cabin_3']] = test_cat_df['Cabin'].str.split('/', expand=True)\ntest_cat_df[['id_1', 'id_2']] = test_cat_df['PassengerId'].str.split('_', expand=True)\ntest_cat_df = test_cat_df.drop(['Cabin', 'PassengerId'], axis = 1)\ntest_cat_df.head(n=10)","metadata":{"execution":{"iopub.status.busy":"2022-07-25T13:23:23.545225Z","iopub.status.idle":"2022-07-25T13:23:23.545648Z","shell.execute_reply.started":"2022-07-25T13:23:23.545408Z","shell.execute_reply":"2022-07-25T13:23:23.545425Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Feature Engineering P2 - Gender\n\nOne other feature that we can work to refine is name; given that name is closely correlated with gender which could be correlated with Transporation success we'll use nltk to extract gender from name and see if its correlated with Transportaiton success or failure.\n\nThe hope here is to determine if one gender had a higher transportation rate than the other and capture this in our model if so.","metadata":{}},{"cell_type":"code","source":"def gender_features(word):\n    return {'last_letter':word[-1]}\n\ndef return_gender(input_series, classifier):\n    \n    #Applies classifier to series of names\n    gender = []\n    for i in input_series:\n        gender.append(classifier.classify(gender_features(i)))\n    return gender\n\n# preparing a list of examples and corresponding class labels.\nlabeled_names = ([(name, 'male') for name in names.words('male.txt')]+\n             [(name, 'female') for name in names.words('female.txt')])\n  \nrandom.shuffle(labeled_names)\n  \n# we use the feature extractor to process the names data.\nfeaturesets = [(gender_features(n), gender) \n               for (n, gender)in labeled_names]\n  \n# Divide the resulting list of feature\n# sets into a training set and a test set.\ntrain_set, test_set = featuresets[500:], featuresets[:500]\n  \n# training a new \"naive Bayes\" classifier.\nclassifier = nltk.NaiveBayesClassifier.train(train_set)\n\n\n#Applying classifier on test and train set\ntest_cat_df['Gender'] = return_gender(test_cat_df['Name'], classifier)\ncat_df['Gender'] = return_gender(cat_df['Name'], classifier)","metadata":{"execution":{"iopub.status.busy":"2022-07-25T13:23:23.547177Z","iopub.status.idle":"2022-07-25T13:23:23.547574Z","shell.execute_reply.started":"2022-07-25T13:23:23.547361Z","shell.execute_reply":"2022-07-25T13:23:23.547377Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Visualizing Transportation vs. Gender\nfor_graphing = pd.concat([cat_df, df['Transported']], axis = 1)\nsns.set(rc = {'figure.figsize':(15,12)})\nsns.countplot(data = for_graphing, x = 'Gender', hue = 'Transported').set(title = 'Gender Impact on Transportation')\n\n#Encoding gender as dummy variable\ncat_df = pd.concat([cat_df, pd.get_dummies(cat_df, columns = ['Gender'])], axis = 1).drop(['Gender'], axis = 1)\ntest_cat_df = pd.concat([test_cat_df, pd.get_dummies(test_cat_df, columns = ['Gender'])], axis = 1).drop(['Gender'], axis = 1)","metadata":{"execution":{"iopub.status.busy":"2022-07-25T13:23:23.548778Z","iopub.status.idle":"2022-07-25T13:23:23.549137Z","shell.execute_reply.started":"2022-07-25T13:23:23.548958Z","shell.execute_reply":"2022-07-25T13:23:23.548975Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Looking at the above barplot we can see that there was a greater chance of being succesfully transported for women than men. Although the difference is small we'll include this gender feature in our model.","metadata":{}},{"cell_type":"markdown","source":"## Feature Engineering P3 - Age + Categorical Encoding\n\nThe final feature we'll add is bucketting age, essentially adding a new binary feature for every 10 years and classifying passengers into one of these categories.\n\nThis approach will allow us to capture any non linear relationships between age and Transportation chance.\n\nAfter this we'll perform categorical encoding; one of the requirements of an XGBoosted model is that the features are all numeric. To get our categorical data into an acceptable format we'll use Sklearns one hot encoder to split feature levels into separate binary columns of 0-1.\n\nThis has the potential to significantly expand our dataset so to prevent overfitting we'll drop columns categorical columns with to many levels ( ex. Name).","metadata":{}},{"cell_type":"code","source":"#Bucketting age\ncat_df['age_bucket'] = round(df['Age']/10)\ntest_cat_df['age_bucket'] = round(test_df['Age']/10)\ntest = cat_df.copy()\ntest['Transported'] = df['Transported']\n#Encoding categorical variables\nsns.set(rc = {'figure.figsize':(15,12)})\nencoded_columns = ['HomePlanet', 'CryoSleep', 'Destination', 'VIP', 'cabin_1', 'cabin_3','age_bucket']\nsns.countplot(data = test, x = 'age_bucket', hue = 'Transported').set(title = 'Age Bucket Versus Transportation')\n\nfor column in encoded_columns:\n    tempdf = pd.get_dummies(cat_df[column], prefix=column)\n    tempdf_test = pd.get_dummies(test_cat_df[column], prefix=column)\n    \n    cat_df = pd.merge(\n        left=cat_df,\n        right=tempdf,\n        left_index=True,\n        right_index=True,\n    )\n    \n    test_cat_df= pd.merge(\n        left=test_cat_df,\n        right=tempdf_test,\n        left_index=True,\n        right_index=True,\n    )\n    cat_df = cat_df.drop(columns=column)\n    test_cat_df = test_cat_df.drop(columns=column)\n    \n#cat_df.head(n = 10)\n#test_cat_df.head(n = 10)","metadata":{"execution":{"iopub.status.busy":"2022-07-25T13:23:23.550646Z","iopub.status.idle":"2022-07-25T13:23:23.551613Z","shell.execute_reply.started":"2022-07-25T13:23:23.551287Z","shell.execute_reply":"2022-07-25T13:23:23.551317Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Examining Transportation chance by Age in the above Barchart we can see that passengers in age group 1 and 2 (corresponding to ages 0-10 and 10-20) had a far greater likelihood of being transported than the general population.\n\nThis is what we would expect as in many historical disasters, i.e. the Titanic, there was a preference to giving priority to the youngest passengers.","metadata":{}},{"cell_type":"markdown","source":"## Test Train Split\n\nIn order to evaluate our model's accuracy prior to making predictions on the actual test set we'll now split our training dataset into a train(66%) and evaluation(33%). \n\nThis will allow us to provide an unbiased estimation of model accuracy prior to making final predictions.","metadata":{}},{"cell_type":"code","source":"transformed_df = pd.concat([cat_df, normalized_numeric_df], axis = 1)\ntransformed_df = transformed_df.drop(['Name'], axis = 1)\n\nX_train, X_test, y_train, y_test = train_test_split(transformed_df, df['Transported'], test_size=0.33, random_state=42)\n\n#Transforming test set\ntransformed_test_df = pd.concat([test_cat_df, normalized_numeric_test_df], axis = 1)\ntransformed_test_df = transformed_test_df.drop(['Name'], axis = 1)","metadata":{"execution":{"iopub.status.busy":"2022-07-25T13:23:23.552897Z","iopub.status.idle":"2022-07-25T13:23:23.553420Z","shell.execute_reply.started":"2022-07-25T13:23:23.553144Z","shell.execute_reply":"2022-07-25T13:23:23.553169Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Modeling\n\nWe'll perform 2 rounds of parameter tuning to reduce the training time; first we'll vary min child weight and max depth, second we'll use the best parameters from the first search and vary gamma and subsample","metadata":{}},{"cell_type":"code","source":"params = {\n        'min_child_weight': [1,3, 5],\n        'max_depth': [4,5,6, 7, 8,9],\n        'verbosity': [0]\n    \n        }\nX_train = X_train.apply(pd.to_numeric)\nX_train = X_train.loc[:,~X_train.columns.duplicated()].copy()\nX_train.head()\nxgb = XGBClassifier(learning_rate=0.02, n_estimators=600, objective='binary:logistic',\n                  silent=True, nthread=-1)\n\ngb= GridSearchCV(xgb, params, cv = 3)\ngb.fit(X_train, y_train)\ngb.best_params_","metadata":{"execution":{"iopub.status.busy":"2022-07-25T13:23:23.554914Z","iopub.status.idle":"2022-07-25T13:23:23.555434Z","shell.execute_reply.started":"2022-07-25T13:23:23.555162Z","shell.execute_reply":"2022-07-25T13:23:23.555187Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"grid_scores = gb.cv_results_\ngb_df = pd.DataFrame.from_dict(grid_scores)\ngb_df.head()\nsns.set(rc = {'figure.figsize':(15,12)})\nsns.lineplot(data = gb_df, x = 'param_max_depth', y = 'mean_test_score', hue = 'param_min_child_weight')","metadata":{"execution":{"iopub.status.busy":"2022-07-25T13:23:23.557025Z","iopub.status.idle":"2022-07-25T13:23:23.557565Z","shell.execute_reply.started":"2022-07-25T13:23:23.557283Z","shell.execute_reply":"2022-07-25T13:23:23.557308Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Second round of tuning\nparams = {\n        'min_child_weight': [3],\n        'gamma': [ 1, 1.5, 2],\n        'subsample': [0.6, 0.8, 1.0, 1.2],\n        'max_depth': [5],\n        'verbosity': [0],\n        'min_child_weight' :[5]\n    \n        }\n#X_train = X_train.apply(pd.to_numeric)\n#X_train = X_train.loc[:,~X_train.columns.duplicated()].copy()\n#X_train.head()\nxgb = XGBClassifier(learning_rate=0.02, n_estimators=600, objective='binary:logistic',\n                  silent=True, nthread=-1)\n\ngb= GridSearchCV(xgb, params, cv = 3)\ngb.fit(X_train, y_train)\ngb.best_params_","metadata":{"execution":{"iopub.status.busy":"2022-07-25T13:23:23.558789Z","iopub.status.idle":"2022-07-25T13:23:23.559316Z","shell.execute_reply.started":"2022-07-25T13:23:23.559049Z","shell.execute_reply":"2022-07-25T13:23:23.559074Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"grid_scores = gb.cv_results_\ngb_df = pd.DataFrame.from_dict(grid_scores)\ngb_df.head()\nsns.set(rc = {'figure.figsize':(15,12)})\nsns.lineplot(data = gb_df, x = 'param_gamma', y = 'mean_test_score', hue = 'param_subsample')","metadata":{"execution":{"iopub.status.busy":"2022-07-25T13:23:23.561196Z","iopub.status.idle":"2022-07-25T13:23:23.561732Z","shell.execute_reply.started":"2022-07-25T13:23:23.561442Z","shell.execute_reply":"2022-07-25T13:23:23.561467Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Training Final Model\nUsing the parameters identified above we'll retrain our model on the entire dataset and use it to predict results.","metadata":{}},{"cell_type":"code","source":"#Final model\ntransformed_df = transformed_df .loc[:,~transformed_df .columns.duplicated()].copy()\ntransformed_df = transformed_df.apply(pd.to_numeric)\ncomplete_model_xgb = XGBClassifier(max_depth = 5, gamma = 2,min_child_weight = 5,subsample = 1, n_jobs = -1, verbosity = 0, eval_metric = 'mlogloss')\ncomplete_model_xgb.fit(transformed_df, df['Transported'])\nfinal_scores = cross_val_score(complete_model_xgb, transformed_df, df['Transported'], cv=10)\n#print(final_scores)\n\n#Plotting feature importance\nplot_importance(complete_model_xgb)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-25T13:23:23.563940Z","iopub.status.idle":"2022-07-25T13:23:23.564454Z","shell.execute_reply.started":"2022-07-25T13:23:23.564179Z","shell.execute_reply":"2022-07-25T13:23:23.564205Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Submitting Results","metadata":{}},{"cell_type":"code","source":"transformed_test_df = transformed_test_df .loc[:,~transformed_test_df .columns.duplicated()].copy()\ntransformed_test_df = transformed_test_df.apply(pd.to_numeric)\nstarship_prediction_xg = complete_model_xgb.predict(transformed_test_df)\n\nlabels = test_df['PassengerId']\nstarship_submission = pd.DataFrame(np.array([labels, starship_prediction_xg]).T,\n                                 columns = ['PassengerId', 'Transported'])\n#Converting predictions to True/False\nstarship_submission['Transported'] = starship_submission['Transported'].replace(0, 'False')\nstarship_submission['Transported'] = starship_submission['Transported'].replace(1, 'True')\n\n#Submitting\nstarship_submission.to_csv('submission2.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-25T13:23:23.569058Z","iopub.status.idle":"2022-07-25T13:23:23.569484Z","shell.execute_reply.started":"2022-07-25T13:23:23.569282Z","shell.execute_reply":"2022-07-25T13:23:23.569305Z"},"trusted":true},"execution_count":null,"outputs":[]}]}