{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-18T12:34:02.080730Z","iopub.execute_input":"2022-07-18T12:34:02.081510Z","iopub.status.idle":"2022-07-18T12:34:02.118229Z","shell.execute_reply.started":"2022-07-18T12:34:02.081365Z","shell.execute_reply":"2022-07-18T12:34:02.116889Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Imports\nimport operator as op\n\n# sklearn\nfrom sklearn.preprocessing import LabelEncoder, OrdinalEncoder\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.model_selection import train_test_split, GridSearchCV, StratifiedKFold\nfrom sklearn.metrics import accuracy_score, f1_score, precision_score, recall_score, classification_report, mean_squared_error, plot_confusion_matrix\n\n# stats models\nfrom statsmodels.stats.outliers_influence import variance_inflation_factor\n\n# Visualization\nimport matplotlib.pyplot as plt\n%matplotlib inline\nimport seaborn as sns\n\n# XGBoost\nfrom xgboost import XGBClassifier","metadata":{"execution":{"iopub.status.busy":"2022-07-18T12:36:32.691969Z","iopub.execute_input":"2022-07-18T12:36:32.692362Z","iopub.status.idle":"2022-07-18T12:36:33.864811Z","shell.execute_reply.started":"2022-07-18T12:36:32.692330Z","shell.execute_reply":"2022-07-18T12:36:33.863575Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Import the data\ntrain = pd.read_csv('../input/spaceship-titanic/train.csv')\ntest = pd.read_csv('../input/spaceship-titanic/test.csv')","metadata":{"execution":{"iopub.status.busy":"2022-07-18T13:14:16.211806Z","iopub.execute_input":"2022-07-18T13:14:16.212215Z","iopub.status.idle":"2022-07-18T13:14:16.264057Z","shell.execute_reply.started":"2022-07-18T13:14:16.212184Z","shell.execute_reply":"2022-07-18T13:14:16.263171Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Check sizes\nprint(\"Train shape: {}\".format(train.shape))\nprint(\"Test shape: {}\".format(test.shape))","metadata":{"execution":{"iopub.status.busy":"2022-07-18T13:14:16.864885Z","iopub.execute_input":"2022-07-18T13:14:16.865327Z","iopub.status.idle":"2022-07-18T13:14:16.872319Z","shell.execute_reply.started":"2022-07-18T13:14:16.865293Z","shell.execute_reply":"2022-07-18T13:14:16.870885Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# View train\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-18T13:14:17.635789Z","iopub.execute_input":"2022-07-18T13:14:17.636588Z","iopub.status.idle":"2022-07-18T13:14:17.660340Z","shell.execute_reply.started":"2022-07-18T13:14:17.636545Z","shell.execute_reply":"2022-07-18T13:14:17.659199Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Describe the training set\ntrain.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-18T13:14:21.271619Z","iopub.execute_input":"2022-07-18T13:14:21.272640Z","iopub.status.idle":"2022-07-18T13:14:21.295203Z","shell.execute_reply.started":"2022-07-18T13:14:21.272572Z","shell.execute_reply":"2022-07-18T13:14:21.293986Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Describe test set\ntest.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-18T13:14:22.121568Z","iopub.execute_input":"2022-07-18T13:14:22.121970Z","iopub.status.idle":"2022-07-18T13:14:22.139124Z","shell.execute_reply.started":"2022-07-18T13:14:22.121937Z","shell.execute_reply":"2022-07-18T13:14:22.138112Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Unique values\nfor col in train.columns:\n    if (train[col].dtype == 'object'):\n        print(f\"{col}: {train[col].unique()}\")","metadata":{"execution":{"iopub.status.busy":"2022-07-18T13:14:24.062219Z","iopub.execute_input":"2022-07-18T13:14:24.063048Z","iopub.status.idle":"2022-07-18T13:14:24.079471Z","shell.execute_reply.started":"2022-07-18T13:14:24.062993Z","shell.execute_reply":"2022-07-18T13:14:24.078045Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Test unique values\nfor col in test.columns:\n    if (test[col].dtype == 'object'):\n        print(f\"{col}: {test[col].unique()}\")","metadata":{"execution":{"iopub.status.busy":"2022-07-18T13:14:24.783241Z","iopub.execute_input":"2022-07-18T13:14:24.783954Z","iopub.status.idle":"2022-07-18T13:14:24.794901Z","shell.execute_reply.started":"2022-07-18T13:14:24.783914Z","shell.execute_reply":"2022-07-18T13:14:24.793985Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Null values\ntrain.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2022-07-18T13:14:25.605636Z","iopub.execute_input":"2022-07-18T13:14:25.606217Z","iopub.status.idle":"2022-07-18T13:14:25.625469Z","shell.execute_reply.started":"2022-07-18T13:14:25.606181Z","shell.execute_reply":"2022-07-18T13:14:25.624132Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Test null values\ntest.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2022-07-18T13:14:27.806965Z","iopub.execute_input":"2022-07-18T13:14:27.807742Z","iopub.status.idle":"2022-07-18T13:14:27.823807Z","shell.execute_reply.started":"2022-07-18T13:14:27.807702Z","shell.execute_reply":"2022-07-18T13:14:27.822837Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Train duplicate check\ntrain.duplicated(keep=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-18T13:14:28.435449Z","iopub.execute_input":"2022-07-18T13:14:28.436306Z","iopub.status.idle":"2022-07-18T13:14:28.461381Z","shell.execute_reply.started":"2022-07-18T13:14:28.436250Z","shell.execute_reply":"2022-07-18T13:14:28.460266Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Test duplicate check\ntest.duplicated(keep=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-18T13:14:29.120400Z","iopub.execute_input":"2022-07-18T13:14:29.121351Z","iopub.status.idle":"2022-07-18T13:14:29.138492Z","shell.execute_reply.started":"2022-07-18T13:14:29.121275Z","shell.execute_reply":"2022-07-18T13:14:29.136986Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Target variable\ntarget = train.Transported","metadata":{"execution":{"iopub.status.busy":"2022-07-18T13:14:29.852711Z","iopub.execute_input":"2022-07-18T13:14:29.853497Z","iopub.status.idle":"2022-07-18T13:14:29.859168Z","shell.execute_reply.started":"2022-07-18T13:14:29.853448Z","shell.execute_reply":"2022-07-18T13:14:29.858031Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Label Encoder\nenc = LabelEncoder()\n\n# Fit to the target\ny = enc.fit_transform(target)","metadata":{"execution":{"iopub.status.busy":"2022-07-18T13:14:30.861954Z","iopub.execute_input":"2022-07-18T13:14:30.862326Z","iopub.status.idle":"2022-07-18T13:14:30.867162Z","shell.execute_reply.started":"2022-07-18T13:14:30.862297Z","shell.execute_reply":"2022-07-18T13:14:30.866334Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Retain PassengerId separately for submitting predictions\npassenger_id = test.PassengerId","metadata":{"execution":{"iopub.status.busy":"2022-07-18T13:14:35.387470Z","iopub.execute_input":"2022-07-18T13:14:35.387837Z","iopub.status.idle":"2022-07-18T13:14:35.393471Z","shell.execute_reply.started":"2022-07-18T13:14:35.387808Z","shell.execute_reply":"2022-07-18T13:14:35.392139Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Previous iterations of work on this competition revealed an overfitted model that stemmed from problems in the construction of the model itself. This notebook will explore more feature engineering techniques to try and improve the model that is created.","metadata":{}},{"cell_type":"markdown","source":"# Missing Data\n\nFor categorical data, it is best to use the most frequent response to replace missing data. For continuous variables, like Age or RoomService, the choice comes down to mean or median. This choice is best made based on the distribution of data.","metadata":{}},{"cell_type":"markdown","source":"## Continuous Variables","metadata":{}},{"cell_type":"code","source":"# View continuous variable distributions\nfig, ((ax0, ax1, ax2), (ax3, ax4, ax5)) = plt.subplots(2, 3, figsize=(20,20))\n\n# Title\nfig.suptitle('Continuous Variables')\nax0.set_title('Age')\nax1.set_title('Room Service')\nax2.set_title('Food Court')\nax3.set_title('Shopping Mall')\nax4.set_title('Spa')\nax5.set_title('VR Deck')\n\n# Visualizations\nax0 = sns.kdeplot(x = 'Age', data = train, hue = 'Transported', ax = ax0)\nax1 = sns.kdeplot(x = 'RoomService', data = train, hue = 'Transported', ax = ax1)\nax2 = sns.kdeplot(x = 'FoodCourt', data = train, hue = 'Transported', ax = ax2)\nax3 = sns.kdeplot(x = 'ShoppingMall', data = train, hue = 'Transported', ax = ax3)\nax4 = sns.kdeplot(x = 'Spa', data = train, hue = 'Transported', ax = ax4)\nax5 = sns.kdeplot(x = 'VRDeck', data = train, hue = 'Transported', ax = ax5)","metadata":{"execution":{"iopub.status.busy":"2022-07-18T13:14:37.641492Z","iopub.execute_input":"2022-07-18T13:14:37.641897Z","iopub.status.idle":"2022-07-18T13:14:39.025205Z","shell.execute_reply.started":"2022-07-18T13:14:37.641843Z","shell.execute_reply":"2022-07-18T13:14:39.023955Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Boxplots\nfig, ((ax0, ax1, ax2), (ax3, ax4, ax5)) = plt.subplots(2, 3, figsize=(20,20))\n\n# Title\nfig.suptitle('Continuous Variables')\nax0.set_title('Age')\nax1.set_title('Room Service')\nax2.set_title('Food Court')\nax3.set_title('Shopping Mall')\nax4.set_title('Spa')\nax5.set_title('VR Deck')\n\n# Visualizations\nax0 = sns.boxplot(x = 'Age', data = train, ax = ax0)\nax1 = sns.boxplot(x = 'RoomService', data = train, ax = ax1)\nax2 = sns.boxplot(x = 'FoodCourt', data = train, ax = ax2)\nax3 = sns.boxplot(x = 'ShoppingMall', data = train, ax = ax3)\nax4 = sns.boxplot(x = 'Spa', data = train, ax = ax4)\nax5 = sns.boxplot(x = 'VRDeck', data = train, ax = ax5)","metadata":{"execution":{"iopub.status.busy":"2022-07-18T13:14:43.301767Z","iopub.execute_input":"2022-07-18T13:14:43.302536Z","iopub.status.idle":"2022-07-18T13:14:43.955225Z","shell.execute_reply.started":"2022-07-18T13:14:43.302497Z","shell.execute_reply":"2022-07-18T13:14:43.954440Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Age has a close to normal distribution, and given the sample size it's OK to use the mean as a measure of central tendency. The rest, however, show a significant positive skew and as such should use the median value. Furthermore, these variables will each be replaced by two variables:\n- Missingness indicator: If the original value is 0, this column will indicate a 1 for true. If non-zero, it will indicate false with a 0.\n- Log(x+1) transformation of the base value.\n\nLooking at the boxplots,there are a significant set of outliers for several of the variables, it seems:\n- RoomService\n- FoodCourt\n- ShoppingMall\n- Spa\n- VRDeck\n\nThese, along with any zero values, can be handled with transformations.","metadata":{}},{"cell_type":"markdown","source":"## Categorical Variables","metadata":{}},{"cell_type":"code","source":"# Countplots\nfig, (ax0, ax1, ax2, ax3, ax4) = plt.subplots(1, 5, figsize=(30,15))\n\n# Title\nfig.suptitle('Categorical Variables')\nax0.set_title('HomePlanet')\nax1.set_title('CryoSleep')\nax2.set_title('Cabin')\nax3.set_title('Destination')\nax4.set_title('VIP')\n\n# Visualizations\nax0 = sns.countplot(x = 'HomePlanet', data = train, ax = ax0)\nax1 = sns.countplot(x = 'CryoSleep', data = train, ax = ax1)\nax2 = sns.countplot(x = 'Cabin', data = train, ax = ax2)\nax3 = sns.countplot(x = 'Destination', data = train, ax = ax3)\nax4 = sns.countplot(x = 'VIP', data = train, ax = ax4)","metadata":{"execution":{"iopub.status.busy":"2022-07-18T13:23:39.364274Z","iopub.execute_input":"2022-07-18T13:23:39.364687Z","iopub.status.idle":"2022-07-18T13:24:53.418384Z","shell.execute_reply.started":"2022-07-18T13:23:39.364656Z","shell.execute_reply":"2022-07-18T13:24:53.417254Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Dependent variable\nplt.figure(figsize=(14,14))\n\n# barchart\nsns.countplot(x = 'Transported', data = train)\n\n# Show\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-18T13:27:03.895825Z","iopub.execute_input":"2022-07-18T13:27:03.896362Z","iopub.status.idle":"2022-07-18T13:27:04.078936Z","shell.execute_reply.started":"2022-07-18T13:27:03.896315Z","shell.execute_reply":"2022-07-18T13:27:04.077651Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Columns with missing values\ntrain_nan_cols = train.columns[train.isnull().any()]\n\n# Replace missing values in Age with mean\ntrain.Age.fillna(train.Age.mean(), inplace=True)\n\n# Replace missing values in RoomService, FoodCourt, ShoppingMall, Spa, and VRDeck with median\n# Replace categorical values with most common values\nfor col in train_nan_cols:\n    if train[col].dtype == 'object':\n        train[col].fillna(train[col].mode().iloc[0], inplace=True)\n    elif train[col].dtype == 'float64':\n        if train[col].name != 'Age':\n            train[col].fillna(train[col].median(), inplace=True)\n        else:\n            # Do nothing\n            continue","metadata":{"execution":{"iopub.status.busy":"2022-07-18T13:28:06.136564Z","iopub.execute_input":"2022-07-18T13:28:06.137008Z","iopub.status.idle":"2022-07-18T13:28:06.183715Z","shell.execute_reply.started":"2022-07-18T13:28:06.136975Z","shell.execute_reply":"2022-07-18T13:28:06.182671Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Check that there are no null values\ntrain.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2022-07-18T13:28:06.960561Z","iopub.execute_input":"2022-07-18T13:28:06.961400Z","iopub.status.idle":"2022-07-18T13:28:06.977441Z","shell.execute_reply.started":"2022-07-18T13:28:06.961362Z","shell.execute_reply":"2022-07-18T13:28:06.976591Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Columns with missing values\ntest_nan_cols = test.columns[test.isnull().any()]\n\n# Replace missing values in Age with mean\ntest.Age.fillna(test.Age.mean(), inplace=True)\n\n# Replace missing values in RoomService, FoodCourt, ShoppingMall, Spa, and VRDeck with median\n# Replace categorical values with most common values\nfor col in test_nan_cols:\n    if test[col].dtype == 'object':\n        test[col].fillna(test[col].mode().iloc[0], inplace=True)\n    elif test[col].dtype == 'float64':\n        if test[col].name != 'Age':\n            test[col].fillna(test[col].median(), inplace=True)\n        else:\n            # Do nothing\n            continue","metadata":{"execution":{"iopub.status.busy":"2022-07-18T13:28:10.155637Z","iopub.execute_input":"2022-07-18T13:28:10.156048Z","iopub.status.idle":"2022-07-18T13:28:10.187570Z","shell.execute_reply.started":"2022-07-18T13:28:10.156013Z","shell.execute_reply":"2022-07-18T13:28:10.186306Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Check that there are no null values\ntest.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2022-07-18T13:28:10.695258Z","iopub.execute_input":"2022-07-18T13:28:10.695650Z","iopub.status.idle":"2022-07-18T13:28:10.709926Z","shell.execute_reply.started":"2022-07-18T13:28:10.695616Z","shell.execute_reply":"2022-07-18T13:28:10.708638Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Drop Name from train and test\ntrain.drop(['Name'], axis = 1)\ntest.drop(['Name'], axis = 1)","metadata":{"execution":{"iopub.status.busy":"2022-07-18T13:28:13.050774Z","iopub.execute_input":"2022-07-18T13:28:13.051180Z","iopub.status.idle":"2022-07-18T13:28:13.084555Z","shell.execute_reply.started":"2022-07-18T13:28:13.051143Z","shell.execute_reply":"2022-07-18T13:28:13.083340Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Zero-Value Detection and Transformations\n\nColumns for zero-value detection and transformations are:\n- RoomService\n- FoodCourt\n- ShoppingMall\n- Spa\n- VRDeck","metadata":{}},{"cell_type":"code","source":"# Missingness indicator\n# col_name should be in format df.column\ndef zeroDetector(col_name):\n    # list to hold one hot encoding\n    zero_detect = []\n    # Iterate through column list\n    for i in col_name:\n        if i == 0.0:\n            zero_detect.append(1)\n        elif i > 0.0:\n            zero_detect.append(0)\n        else:\n            print(\"Negative value at {}\".format(col_name.loc[i]))\n    # Return zero_detect\n    return zero_detect","metadata":{"execution":{"iopub.status.busy":"2022-07-18T13:28:16.991580Z","iopub.execute_input":"2022-07-18T13:28:16.992017Z","iopub.status.idle":"2022-07-18T13:28:16.999137Z","shell.execute_reply.started":"2022-07-18T13:28:16.991983Z","shell.execute_reply":"2022-07-18T13:28:16.997797Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Log transform all values\n# col_name should be in format df.column\ndef logTx(col_name):\n    # Hold transformed column values\n    tx_col = np.log(col_name + 1)\n    # Return tx_col\n    return tx_col","metadata":{"execution":{"iopub.status.busy":"2022-07-18T13:28:17.687802Z","iopub.execute_input":"2022-07-18T13:28:17.688593Z","iopub.status.idle":"2022-07-18T13:28:17.694879Z","shell.execute_reply.started":"2022-07-18T13:28:17.688547Z","shell.execute_reply":"2022-07-18T13:28:17.693815Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Training Set","metadata":{}},{"cell_type":"code","source":"# RoomService\ntrain_roomservice_zeroes = zeroDetector(train.RoomService)\ntrain_roomservice_tfmd = logTx(train.RoomService)\n\n# FoodCourt\ntrain_foodcourt_zeroes = zeroDetector(train.FoodCourt)\ntrain_foodcourt_tfmd = logTx(train.FoodCourt)\n\n# ShoppingMall\ntrain_shoppingmall_zeroes = zeroDetector(train.ShoppingMall)\ntrain_shoppingmall_tfmd = logTx(train.ShoppingMall)\n\n# Spa\ntrain_spa_zeroes = zeroDetector(train.Spa)\ntrain_spa_tfmd = logTx(train.Spa)\n\n# VRDeck\ntrain_vrdeck_zeroes = zeroDetector(train.VRDeck)\ntrain_vrdeck_tfmd = logTx(train.VRDeck)","metadata":{"execution":{"iopub.status.busy":"2022-07-18T13:28:20.606524Z","iopub.execute_input":"2022-07-18T13:28:20.607008Z","iopub.status.idle":"2022-07-18T13:28:20.631664Z","shell.execute_reply.started":"2022-07-18T13:28:20.606974Z","shell.execute_reply":"2022-07-18T13:28:20.630272Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Testing Set","metadata":{}},{"cell_type":"code","source":"# RoomService\ntest_roomservice_zeroes = zeroDetector(test.RoomService)\ntest_roomservice_tfmd = logTx(test.RoomService)\n\n# FoodCourt\ntest_foodcourt_zeroes = zeroDetector(test.FoodCourt)\ntest_foodcourt_tfmd = logTx(test.FoodCourt)\n\n# ShoppingMall\ntest_shoppingmall_zeroes = zeroDetector(test.ShoppingMall)\ntest_shoppingmall_tfmd = logTx(test.ShoppingMall)\n\n# Spa\ntest_spa_zeroes = zeroDetector(test.Spa)\ntest_spa_tfmd = logTx(test.Spa)\n\n# VRDeck\ntest_vrdeck_zeroes = zeroDetector(test.VRDeck)\ntest_vrdeck_tfmd = logTx(test.VRDeck)","metadata":{"execution":{"iopub.status.busy":"2022-07-18T13:28:21.870800Z","iopub.execute_input":"2022-07-18T13:28:21.871762Z","iopub.status.idle":"2022-07-18T13:28:21.889901Z","shell.execute_reply.started":"2022-07-18T13:28:21.871715Z","shell.execute_reply":"2022-07-18T13:28:21.888662Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## PassengerId\n\nPassengerId has two components separated by an underscore. The first component is a group identifier, and the second component is counter to describe how many travelers are in the identified group. This variable can therefore be split and used to create a new variable that tracks how many traveling companions an individual traveler has.","metadata":{}},{"cell_type":"code","source":"# Split PassengerId into its component parts and count the number of traveling companions per traveler\n# p_id_col is a column in a dataframe. Pass it as df.column\ndef passengerIdSplitter(p_id_col):\n    # Split PassengerID to find traveling companions\n    groups = [i.split('_')[0] for i in p_id_col]\n    # Hold the companion counts\n    companion_count = []\n    # Count the companions\n    for p in p_id_col:\n        # Reset\n        count = 0\n        # Split to get the group id\n        group = p.split('_')[0]\n        # Count instances - 1 for companion count\n        count = op.countOf(groups, group)-1\n        # Add to list\n        companion_count.append(count)\n    # Return full list of counts\n    return companion_count","metadata":{"execution":{"iopub.status.busy":"2022-07-18T13:28:25.973265Z","iopub.execute_input":"2022-07-18T13:28:25.974491Z","iopub.status.idle":"2022-07-18T13:28:25.982047Z","shell.execute_reply.started":"2022-07-18T13:28:25.974443Z","shell.execute_reply":"2022-07-18T13:28:25.980826Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Training Set","metadata":{}},{"cell_type":"code","source":"# Get companion count for PassengerId in train data set\ntrain_companion_count = passengerIdSplitter(train.PassengerId)","metadata":{"execution":{"iopub.status.busy":"2022-07-18T13:28:29.326255Z","iopub.execute_input":"2022-07-18T13:28:29.326618Z","iopub.status.idle":"2022-07-18T13:28:31.867424Z","shell.execute_reply.started":"2022-07-18T13:28:29.326590Z","shell.execute_reply":"2022-07-18T13:28:31.866342Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Testing Set","metadata":{}},{"cell_type":"code","source":"# Get companion count for PassengerId in train data set\ntest_companion_count = passengerIdSplitter(test.PassengerId)","metadata":{"execution":{"iopub.status.busy":"2022-07-18T13:28:31.869626Z","iopub.execute_input":"2022-07-18T13:28:31.870099Z","iopub.status.idle":"2022-07-18T13:28:32.496505Z","shell.execute_reply.started":"2022-07-18T13:28:31.870056Z","shell.execute_reply":"2022-07-18T13:28:32.495483Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Cabin\n\nCabin will be split into 3 levels:\n- deck\n- level\n- side","metadata":{}},{"cell_type":"code","source":"# Split cabin into deck, level, and side\n# cabin_col should be entered in the format df.cabin\ndef cabinSplitter(cabin_col):\n    # Create the lists to house the variables\n    deck = []\n    level = []\n    side = []\n    # Create deck, level, and room type\n    for c in cabin_col:\n        # If there is a null value\n        if c is np.nan:\n            deck.append(np.nan)\n            level.append(np.nan)\n            side.append(np.nan)\n        # If not null\n        else:\n            c_split = str(c).split('/', maxsplit=-1)\n            deck.append(c_split[0])\n            level.append(c_split[1])\n            side.append(c_split[2])\n    # Return\n    return deck, level, side","metadata":{"execution":{"iopub.status.busy":"2022-07-18T13:28:33.780484Z","iopub.execute_input":"2022-07-18T13:28:33.781113Z","iopub.status.idle":"2022-07-18T13:28:33.790506Z","shell.execute_reply.started":"2022-07-18T13:28:33.781081Z","shell.execute_reply":"2022-07-18T13:28:33.789200Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Training Set","metadata":{}},{"cell_type":"code","source":"# Get deck, level, and side for train dataframe\ntrain_deck, train_level, train_side = cabinSplitter(train.Cabin)","metadata":{"execution":{"iopub.status.busy":"2022-07-18T13:28:35.321773Z","iopub.execute_input":"2022-07-18T13:28:35.322169Z","iopub.status.idle":"2022-07-18T13:28:35.336520Z","shell.execute_reply.started":"2022-07-18T13:28:35.322138Z","shell.execute_reply":"2022-07-18T13:28:35.335207Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Add to the training data set\ntrain.insert(4, 'deck', train_deck)\ntrain.insert(5, 'level', train_level)\ntrain.insert(6, 'side', train_side)\n\n# Check\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-18T13:28:35.941520Z","iopub.execute_input":"2022-07-18T13:28:35.941928Z","iopub.status.idle":"2022-07-18T13:28:35.973462Z","shell.execute_reply.started":"2022-07-18T13:28:35.941896Z","shell.execute_reply":"2022-07-18T13:28:35.972221Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Testing Set","metadata":{}},{"cell_type":"code","source":"# Get deck, level, and side for test dataframe\ntest_deck, test_level, test_side = cabinSplitter(test.Cabin)","metadata":{"execution":{"iopub.status.busy":"2022-07-18T13:28:38.332029Z","iopub.execute_input":"2022-07-18T13:28:38.332410Z","iopub.status.idle":"2022-07-18T13:28:38.342452Z","shell.execute_reply.started":"2022-07-18T13:28:38.332382Z","shell.execute_reply":"2022-07-18T13:28:38.341233Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Add to the training data set\ntest.insert(4, 'deck', test_deck)\ntest.insert(5, 'level', test_level)\ntest.insert(6, 'side', test_side)\n\n# Check\ntest.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-18T13:28:38.932022Z","iopub.execute_input":"2022-07-18T13:28:38.932400Z","iopub.status.idle":"2022-07-18T13:28:38.960197Z","shell.execute_reply.started":"2022-07-18T13:28:38.932371Z","shell.execute_reply":"2022-07-18T13:28:38.959010Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Level","metadata":{}},{"cell_type":"markdown","source":"According to the level column, the highest value is 1894, and the lowest is 0. New categories will be created off of these values as follows:\n- 0 to 400\n- 401 to 800\n- 801 to 1200\n- 1201 to 1600\n- 1601 to 2000","metadata":{}},{"cell_type":"code","source":"# Create categories of levels\n# level_col should be entered as df.level\ndef levelCategorizer(level_col):\n    # Create a list categorizing each level category\n    level_cat = []\n    # Iterate through each value in the level column\n    for l in level_col:\n        if isinstance(l, str):\n            # Convert l into into\n            l = int(l)\n            if l < 401:\n                level_cat.append('<=400')\n            elif l < 801:\n                level_cat.append('401 - 800')\n            elif l < 1201:\n                level_cat.append('801 - 1200')\n            elif l < 1601:\n                level_cat.append('1201 - 1600')\n            elif l < 2001:\n                level_cat.append('1601 - 2000')\n            else:\n                # Do nothing\n                continue\n        else:\n            level_cat.append(np.nan)\n    # Return the level categories\n    return level_cat","metadata":{"execution":{"iopub.status.busy":"2022-07-18T13:28:41.768046Z","iopub.execute_input":"2022-07-18T13:28:41.768450Z","iopub.status.idle":"2022-07-18T13:28:41.777660Z","shell.execute_reply.started":"2022-07-18T13:28:41.768422Z","shell.execute_reply":"2022-07-18T13:28:41.776457Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Training Set","metadata":{}},{"cell_type":"code","source":"# Get the categories\ntrain_level_cat = levelCategorizer(train.level)","metadata":{"execution":{"iopub.status.busy":"2022-07-18T13:28:44.898169Z","iopub.execute_input":"2022-07-18T13:28:44.898552Z","iopub.status.idle":"2022-07-18T13:28:44.909467Z","shell.execute_reply.started":"2022-07-18T13:28:44.898520Z","shell.execute_reply":"2022-07-18T13:28:44.908525Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Testing Set","metadata":{}},{"cell_type":"code","source":"# Get the categories\ntest_level_cat = levelCategorizer(test.level)","metadata":{"execution":{"iopub.status.busy":"2022-07-18T13:28:45.890789Z","iopub.execute_input":"2022-07-18T13:28:45.891546Z","iopub.status.idle":"2022-07-18T13:28:45.899214Z","shell.execute_reply.started":"2022-07-18T13:28:45.891507Z","shell.execute_reply":"2022-07-18T13:28:45.898203Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Age Conversion","metadata":{}},{"cell_type":"code","source":"# Age range\nprint(f\"Minimum age: {min(train.Age)}.\")\nprint(f\"Maximum age: {max(train.Age)}.\")","metadata":{"execution":{"iopub.status.busy":"2022-07-18T13:28:47.016074Z","iopub.execute_input":"2022-07-18T13:28:47.016951Z","iopub.status.idle":"2022-07-18T13:28:47.027057Z","shell.execute_reply.started":"2022-07-18T13:28:47.016902Z","shell.execute_reply":"2022-07-18T13:28:47.025583Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create age categories\n# age_col should be provided as df.Age\ndef ageRanger(age_col):\n    # Hold the age ranges\n    age_ranges = []\n    # Set the values\n    for a in age_col:\n        if a <= 10.0:\n            age_ranges.append(\"<=10.0\")\n        elif a < 21.0:\n            age_ranges.append(\"11.0 - 20.0\")\n        elif a < 31.0:\n            age_ranges.append(\"21.0 - 30.0\")\n        elif a < 41.0:\n            age_ranges.append(\"31.0 - 40.0\")\n        elif a < 51.0:\n            age_ranges.append(\"41.0 - 50.0\")\n        elif a < 61.0:\n            age_ranges.append(\"51.0 - 60.0\")\n        elif a < 71.0:\n            age_ranges.append(\"61.0 - 70.0\")\n        elif a < 81.0:\n            age_ranges.append(\"71.0 - 80.0\")\n        else:\n            age_ranges.append(np.nan)\n    # Return the age ranges\n    return age_ranges","metadata":{"execution":{"iopub.status.busy":"2022-07-18T13:28:48.080906Z","iopub.execute_input":"2022-07-18T13:28:48.082173Z","iopub.status.idle":"2022-07-18T13:28:48.091910Z","shell.execute_reply.started":"2022-07-18T13:28:48.082120Z","shell.execute_reply":"2022-07-18T13:28:48.090929Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Training Set","metadata":{}},{"cell_type":"code","source":"# Get age ranges for training data\ntrain_age_ranges = ageRanger(train.Age)","metadata":{"execution":{"iopub.status.busy":"2022-07-18T13:28:50.372010Z","iopub.execute_input":"2022-07-18T13:28:50.372801Z","iopub.status.idle":"2022-07-18T13:28:50.381811Z","shell.execute_reply.started":"2022-07-18T13:28:50.372753Z","shell.execute_reply":"2022-07-18T13:28:50.380575Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Testing Set","metadata":{}},{"cell_type":"code","source":"# Get age ranges for testing data\ntest_age_ranges = ageRanger(test.Age)","metadata":{"execution":{"iopub.status.busy":"2022-07-18T13:28:51.461572Z","iopub.execute_input":"2022-07-18T13:28:51.462667Z","iopub.status.idle":"2022-07-18T13:28:51.469582Z","shell.execute_reply.started":"2022-07-18T13:28:51.462630Z","shell.execute_reply":"2022-07-18T13:28:51.468458Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Drop Variables\n\n__Variables to drop:__\n- PassengerId\n- Cabin\n- level\n- Room Service\n- Food Court\n- Shopping Mall\n- Spa\n- VRDeck\n\n__Variables to add:__\n- companion_count\n- level_range\n- roomservice_zeroes\n- roomservice_tfmd\n- foodcourt_zeroes\n- foodcourt_tfmd\n- shoppingmall_zeroes\n- shoppingmall_tfmd\n- spa_zeroes\n- spa_tfmd\n- vrdeck_zeroes\n- vrdeck_tfmd","metadata":{}},{"cell_type":"markdown","source":"#### Training Set","metadata":{}},{"cell_type":"code","source":"# Drop variables\ntrain.drop(['PassengerId', 'Cabin', 'level', 'RoomService', 'FoodCourt', 'ShoppingMall', 'Spa', 'VRDeck','Name', 'Transported'], axis=1, inplace=True)\n\n# Check\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-18T13:29:01.644151Z","iopub.execute_input":"2022-07-18T13:29:01.644528Z","iopub.status.idle":"2022-07-18T13:29:01.665996Z","shell.execute_reply.started":"2022-07-18T13:29:01.644498Z","shell.execute_reply":"2022-07-18T13:29:01.664897Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Add variables\ntrain.insert(0, 'companion_count', train_companion_count)\ntrain.insert(4, 'level_range', train_level_cat)\n#train.insert(7, 'age_range', train_age_ranges)\ntrain.insert(8, 'roomservice_zeroes', train_roomservice_zeroes)\ntrain.insert(9, 'roomservice_tfmd', train_roomservice_tfmd)\ntrain.insert(10, 'foodcourt_zeroes', train_foodcourt_zeroes)\ntrain.insert(11, 'foodcourt_tfmd', train_foodcourt_tfmd)\ntrain.insert(12, 'shoppingmall_zeroes', train_shoppingmall_zeroes)\ntrain.insert(13, 'shoppingmall_tfmd', train_shoppingmall_tfmd)\ntrain.insert(14, 'spa_zeroes', train_spa_zeroes)\ntrain.insert(15, 'spa_tfmd', train_spa_tfmd)\ntrain.insert(16, 'vrdeck_zeroes', train_vrdeck_zeroes)\ntrain.insert(17, 'vrdeck_tfmd', train_vrdeck_tfmd)","metadata":{"execution":{"iopub.status.busy":"2022-07-18T13:29:05.551594Z","iopub.execute_input":"2022-07-18T13:29:05.552422Z","iopub.status.idle":"2022-07-18T13:29:05.593730Z","shell.execute_reply.started":"2022-07-18T13:29:05.552378Z","shell.execute_reply":"2022-07-18T13:29:05.592473Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Check\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-18T13:29:06.867779Z","iopub.execute_input":"2022-07-18T13:29:06.868213Z","iopub.status.idle":"2022-07-18T13:29:06.891303Z","shell.execute_reply.started":"2022-07-18T13:29:06.868179Z","shell.execute_reply":"2022-07-18T13:29:06.890125Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Testing Set","metadata":{}},{"cell_type":"code","source":"# Drop variables\ntest.drop(['PassengerId', 'Cabin', 'level', 'RoomService', 'FoodCourt', 'ShoppingMall', 'Spa', 'VRDeck','Name'], axis=1, inplace=True)\n\n# Check\ntest.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-18T13:29:08.228342Z","iopub.execute_input":"2022-07-18T13:29:08.229509Z","iopub.status.idle":"2022-07-18T13:29:08.247099Z","shell.execute_reply.started":"2022-07-18T13:29:08.229469Z","shell.execute_reply":"2022-07-18T13:29:08.245253Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Add variables\ntest.insert(0, 'companion_count', test_companion_count)\ntest.insert(4, 'level_range', test_level_cat)\n#test.insert(7, 'age_range', test_age_ranges)\ntest.insert(8, 'roomservice_zeroes', test_roomservice_zeroes)\ntest.insert(9, 'roomservice_tfmd', test_roomservice_tfmd)\ntest.insert(10, 'foodcourt_zeroes', test_foodcourt_zeroes)\ntest.insert(11, 'foodcourt_tfmd', test_foodcourt_tfmd)\ntest.insert(12, 'shoppingmall_zeroes', test_shoppingmall_zeroes)\ntest.insert(13, 'shoppingmall_tfmd', test_shoppingmall_tfmd)\ntest.insert(14, 'spa_zeroes', test_spa_zeroes)\ntest.insert(15, 'spa_tfmd', test_spa_tfmd)\ntest.insert(16, 'vrdeck_zeroes', test_vrdeck_zeroes)\ntest.insert(17, 'vrdeck_tfmd', test_vrdeck_tfmd)","metadata":{"execution":{"iopub.status.busy":"2022-07-18T13:29:08.936732Z","iopub.execute_input":"2022-07-18T13:29:08.937397Z","iopub.status.idle":"2022-07-18T13:29:08.963196Z","shell.execute_reply.started":"2022-07-18T13:29:08.937359Z","shell.execute_reply":"2022-07-18T13:29:08.962162Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Check\ntest.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-18T13:29:12.271603Z","iopub.execute_input":"2022-07-18T13:29:12.271998Z","iopub.status.idle":"2022-07-18T13:29:12.298652Z","shell.execute_reply.started":"2022-07-18T13:29:12.271967Z","shell.execute_reply":"2022-07-18T13:29:12.297473Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Ordinal Encoding\n\nOrdinal encoder to be used on the following variables:\n- HomePlanet\n- CryoSleep\n- deck\n- level_range\n- side\n- Destination\n- VIP","metadata":{}},{"cell_type":"code","source":"# Initialize OrdinalEncoder instance\nord = OrdinalEncoder(dtype = 'int64')","metadata":{"execution":{"iopub.status.busy":"2022-07-18T13:29:13.638061Z","iopub.execute_input":"2022-07-18T13:29:13.638740Z","iopub.status.idle":"2022-07-18T13:29:13.643026Z","shell.execute_reply.started":"2022-07-18T13:29:13.638701Z","shell.execute_reply":"2022-07-18T13:29:13.642078Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Training Set","metadata":{}},{"cell_type":"code","source":"# Encoding variables\ntrain.HomePlanet = ord.fit_transform(train.HomePlanet.to_numpy().reshape(-1,1))\ntrain.CryoSleep = ord.fit_transform(train.CryoSleep.to_numpy().reshape(-1,1))\ntrain.deck = ord.fit_transform(train.deck.to_numpy().reshape(-1,1))\ntrain.level_range = ord.fit_transform(train.level_range.to_numpy().reshape(-1,1))\ntrain.side = ord.fit_transform(train.side.to_numpy().reshape(-1,1))\ntrain.Destination = ord.fit_transform(train.Destination.to_numpy().reshape(-1,1))\n#train.age_range = ord.fit_transform(train.age_range.to_numpy().reshape(-1,1))\ntrain.VIP = ord.fit_transform(train.VIP.to_numpy().reshape(-1,1))\n\n# Check\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-18T13:29:17.093539Z","iopub.execute_input":"2022-07-18T13:29:17.094021Z","iopub.status.idle":"2022-07-18T13:29:17.141276Z","shell.execute_reply.started":"2022-07-18T13:29:17.093985Z","shell.execute_reply":"2022-07-18T13:29:17.140377Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Testing Set","metadata":{"execution":{"iopub.status.busy":"2022-07-16T01:47:01.555800Z","iopub.execute_input":"2022-07-16T01:47:01.556364Z","iopub.status.idle":"2022-07-16T01:47:01.582661Z","shell.execute_reply.started":"2022-07-16T01:47:01.556331Z","shell.execute_reply":"2022-07-16T01:47:01.581337Z"}}},{"cell_type":"code","source":"# Encoding variables\ntest.HomePlanet = ord.fit_transform(test.HomePlanet.to_numpy().reshape(-1,1))\ntest.CryoSleep = ord.fit_transform(test.CryoSleep.to_numpy().reshape(-1,1))\ntest.deck = ord.fit_transform(test.deck.to_numpy().reshape(-1,1))\ntest.level_range = ord.fit_transform(test.level_range.to_numpy().reshape(-1,1))\ntest.side = ord.fit_transform(test.side.to_numpy().reshape(-1,1))\ntest.Destination = ord.fit_transform(test.Destination.to_numpy().reshape(-1,1))\n#test.age_range = ord.fit_transform(test.age_range.to_numpy().reshape(-1,1))\ntest.VIP = ord.fit_transform(test.VIP.to_numpy().reshape(-1,1))\n\n# Check\ntest.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-18T13:29:18.456234Z","iopub.execute_input":"2022-07-18T13:29:18.456957Z","iopub.status.idle":"2022-07-18T13:29:18.494521Z","shell.execute_reply.started":"2022-07-18T13:29:18.456922Z","shell.execute_reply":"2022-07-18T13:29:18.493206Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Exploratory Data Analysis","metadata":{"_kg_hide-input":false}},{"cell_type":"markdown","source":"## Collinearity Check","metadata":{}},{"cell_type":"code","source":"# Include the predicted variable\ntrain['y']=y\n\n# Heatmap\nplt.figure(figsize=(30,30))\n\nsns.heatmap(train.corr(), annot = True, fmt = 'g')\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-18T13:29:23.052358Z","iopub.execute_input":"2022-07-18T13:29:23.052753Z","iopub.status.idle":"2022-07-18T13:29:25.346524Z","shell.execute_reply.started":"2022-07-18T13:29:23.052723Z","shell.execute_reply":"2022-07-18T13:29:25.345432Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Some of these variables have low R values with regards to the target variable, y. \n- companion_count\n- level_range\n- Age\n\nOther variables, particularly the missingness indicators and transformed cost variables, have high multicollinearity. VIF values should be checked just in case.","metadata":{}},{"cell_type":"code","source":"# Calculate the VIF\n# X is a dataframe, target is the dependent variable column's name\ndef calcVIF(X, target=None):\n    # Drop the target variable first\n    if target != None:\n        X.drop([target], axis=1, inplace=True)\n        # Calculating VIF\n        vif = pd.DataFrame()\n        vif[\"variables\"] = X.columns\n        vif[\"VIF\"] = [variance_inflation_factor(X.values, i) for i in range(X.shape[1])]\n    # Proceed if there is no target variable\n    else:\n        # Calculating VIF\n        vif = pd.DataFrame()\n        vif[\"variables\"] = X.columns\n        vif[\"VIF\"] = [variance_inflation_factor(X.values, i) for i in range(X.shape[1])]\n    # Return the vif dataframe\n    return(vif) ","metadata":{"execution":{"iopub.status.busy":"2022-07-18T13:29:37.201435Z","iopub.execute_input":"2022-07-18T13:29:37.201800Z","iopub.status.idle":"2022-07-18T13:29:37.210160Z","shell.execute_reply.started":"2022-07-18T13:29:37.201768Z","shell.execute_reply":"2022-07-18T13:29:37.209232Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# First VIF check\ncalcVIF(train, target='y')","metadata":{"execution":{"iopub.status.busy":"2022-07-18T13:29:38.478042Z","iopub.execute_input":"2022-07-18T13:29:38.479147Z","iopub.status.idle":"2022-07-18T13:29:38.775684Z","shell.execute_reply.started":"2022-07-18T13:29:38.479105Z","shell.execute_reply":"2022-07-18T13:29:38.774538Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Several of the predictor variables have VIF values over 10.0. They are:\n- roomservice_zeroes, 14.09\n- shoppingmall_zeroes, 13.11\n- foodcourt_zeroes, 12.68\n- vrdeck_zeroes, 12.61\n- spa_zeroes, 11.58\n- deck, 10.69\n\nRegardless, the first pass of this model will run with all of these variables included.","metadata":{}},{"cell_type":"code","source":"# Drop unneeded columns\n#train.drop(['companion_count', 'level_range', 'Age'], axis=1, inplace=True)\n\n\n# Check\n#train.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-18T12:49:34.436780Z","iopub.execute_input":"2022-07-18T12:49:34.437200Z","iopub.status.idle":"2022-07-18T12:49:34.457194Z","shell.execute_reply.started":"2022-07-18T12:49:34.437159Z","shell.execute_reply":"2022-07-18T12:49:34.456099Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Drop unneeded columns from test\n#test.drop(['companion_count', 'level_range', 'Age'], axis=1, inplace=True)\n\n# Check\n#test.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-18T12:49:37.740674Z","iopub.execute_input":"2022-07-18T12:49:37.741690Z","iopub.status.idle":"2022-07-18T12:49:37.762423Z","shell.execute_reply.started":"2022-07-18T12:49:37.741650Z","shell.execute_reply":"2022-07-18T12:49:37.761317Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Create Training and Validation Sets","metadata":{}},{"cell_type":"code","source":"# Set split\nx_train, x_validation, y_train, y_validation = train_test_split(train, y, test_size=0.1, random_state=42)","metadata":{"execution":{"iopub.status.busy":"2022-07-18T13:30:42.200600Z","iopub.execute_input":"2022-07-18T13:30:42.201201Z","iopub.status.idle":"2022-07-18T13:30:42.209942Z","shell.execute_reply.started":"2022-07-18T13:30:42.201151Z","shell.execute_reply":"2022-07-18T13:30:42.208730Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Check shape\nprint(f\"x_train shape: {x_train.shape}.\")\nprint(f\"x_validation shape: {x_validation.shape}.\")","metadata":{"execution":{"iopub.status.busy":"2022-07-18T13:30:43.602381Z","iopub.execute_input":"2022-07-18T13:30:43.602787Z","iopub.status.idle":"2022-07-18T13:30:43.608392Z","shell.execute_reply.started":"2022-07-18T13:30:43.602726Z","shell.execute_reply":"2022-07-18T13:30:43.607338Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Evaluation sets for each iteration\neval_sets = [(x_train, y_train), (x_validation, y_validation)]","metadata":{"execution":{"iopub.status.busy":"2022-07-18T13:30:46.178155Z","iopub.execute_input":"2022-07-18T13:30:46.178874Z","iopub.status.idle":"2022-07-18T13:30:46.183049Z","shell.execute_reply.started":"2022-07-18T13:30:46.178817Z","shell.execute_reply":"2022-07-18T13:30:46.182227Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# XGBoost Classifier + GridSearchCV","metadata":{}},{"cell_type":"code","source":"# Initialize XGBClassifier\nxgbc = XGBClassifier(objective= 'binary:logistic',\n                     nthread=6,\n                     seed=42,\n                     eval_metric = 'logloss')","metadata":{"execution":{"iopub.status.busy":"2022-07-18T13:32:06.216785Z","iopub.execute_input":"2022-07-18T13:32:06.217191Z","iopub.status.idle":"2022-07-18T13:32:06.223235Z","shell.execute_reply.started":"2022-07-18T13:32:06.217156Z","shell.execute_reply":"2022-07-18T13:32:06.221866Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Parameter dict for XGBC GridSearchCV object\nparams = {'max_depth': [5, 6, 8, 10, 20],\n          'n_estimators': [100, 200, 500, 1000],\n          'eta': [0.05, 0.1, 0.03, 0.01]}","metadata":{"execution":{"iopub.status.busy":"2022-07-18T13:32:06.866976Z","iopub.execute_input":"2022-07-18T13:32:06.868036Z","iopub.status.idle":"2022-07-18T13:32:06.873961Z","shell.execute_reply.started":"2022-07-18T13:32:06.867992Z","shell.execute_reply":"2022-07-18T13:32:06.872791Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Initialize the GridSearchCV object\ngc1 = GridSearchCV(xgbc, param_grid=params, cv=10, n_jobs=-1)","metadata":{"execution":{"iopub.status.busy":"2022-07-18T13:32:07.916522Z","iopub.execute_input":"2022-07-18T13:32:07.919066Z","iopub.status.idle":"2022-07-18T13:32:07.924374Z","shell.execute_reply.started":"2022-07-18T13:32:07.919008Z","shell.execute_reply":"2022-07-18T13:32:07.923214Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Fit the grid search object\ngc1.fit(x_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2022-07-18T13:32:09.150166Z","iopub.execute_input":"2022-07-18T13:32:09.151225Z","iopub.status.idle":"2022-07-18T14:34:10.204002Z","shell.execute_reply.started":"2022-07-18T13:32:09.151179Z","shell.execute_reply":"2022-07-18T14:34:10.202894Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# To a dataframe\ngc1_results_pd = pd.DataFrame(gc1.cv_results_)\n\n# View\ngc1_results_pd","metadata":{"execution":{"iopub.status.busy":"2022-07-18T14:35:41.939322Z","iopub.execute_input":"2022-07-18T14:35:41.939756Z","iopub.status.idle":"2022-07-18T14:35:41.985668Z","shell.execute_reply.started":"2022-07-18T14:35:41.939718Z","shell.execute_reply":"2022-07-18T14:35:41.983898Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# XGBClassifier Based on Best Parameters","metadata":{}},{"cell_type":"code","source":"# View the best parameters\ngc1.best_params_","metadata":{"execution":{"iopub.status.busy":"2022-07-18T14:35:46.523754Z","iopub.execute_input":"2022-07-18T14:35:46.524154Z","iopub.status.idle":"2022-07-18T14:35:46.531490Z","shell.execute_reply.started":"2022-07-18T14:35:46.524121Z","shell.execute_reply":"2022-07-18T14:35:46.530486Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The above parameters produced a training accuracy of 83.6%, and a validation accuracy of 78.9%. There is room to do better.","metadata":{}},{"cell_type":"code","source":"# Initialize a new XGBClassifier instance\nxgbc1 = XGBClassifier(objective= 'binary:logistic',\n                      max_depth=8,\n                      n_estimators=200,\n                      eta=0.05,\n                      #colsample_bytree=0.8,\n                      #subsample=0.8,\n                      #gamma=1.5,\n                      nthread=4,\n                      seed=42,\n                      eval_metric = ['error','logloss']) ","metadata":{"execution":{"iopub.status.busy":"2022-07-18T14:40:56.565260Z","iopub.execute_input":"2022-07-18T14:40:56.565685Z","iopub.status.idle":"2022-07-18T14:40:56.573294Z","shell.execute_reply.started":"2022-07-18T14:40:56.565653Z","shell.execute_reply":"2022-07-18T14:40:56.571863Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Fit\nxgbc1.fit(x_train, y_train, eval_set = eval_sets)","metadata":{"execution":{"iopub.status.busy":"2022-07-18T14:40:57.439572Z","iopub.execute_input":"2022-07-18T14:40:57.440063Z","iopub.status.idle":"2022-07-18T14:41:00.588686Z","shell.execute_reply.started":"2022-07-18T14:40:57.440026Z","shell.execute_reply":"2022-07-18T14:41:00.587562Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Accuracy score\nxgbc1_val_acc_score = accuracy_score(y_validation, xgbc1.predict(x_validation))\nxgbc1_train_acc_score = accuracy_score(y_train, xgbc1.predict(x_train))\n\n# View\nprint(f\"Validation set accuracy score: {xgbc1_val_acc_score}.\")\nprint(f\"Training set accuracy score: {xgbc1_train_acc_score}.\")","metadata":{"execution":{"iopub.status.busy":"2022-07-18T14:41:10.146704Z","iopub.execute_input":"2022-07-18T14:41:10.147109Z","iopub.status.idle":"2022-07-18T14:41:10.195298Z","shell.execute_reply.started":"2022-07-18T14:41:10.147080Z","shell.execute_reply":"2022-07-18T14:41:10.194359Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# View confusion matrix\nplot_confusion_matrix(xgbc1, x_validation, y_validation, cmap='rocket')","metadata":{"execution":{"iopub.status.busy":"2022-07-18T14:41:14.750532Z","iopub.execute_input":"2022-07-18T14:41:14.751747Z","iopub.status.idle":"2022-07-18T14:41:14.979825Z","shell.execute_reply.started":"2022-07-18T14:41:14.751692Z","shell.execute_reply":"2022-07-18T14:41:14.978921Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Validation classification report\nprint(classification_report(y_validation, xgbc1.predict(x_validation)))","metadata":{"execution":{"iopub.status.busy":"2022-07-18T14:41:15.768447Z","iopub.execute_input":"2022-07-18T14:41:15.769737Z","iopub.status.idle":"2022-07-18T14:41:15.794614Z","shell.execute_reply.started":"2022-07-18T14:41:15.769694Z","shell.execute_reply":"2022-07-18T14:41:15.793698Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Training classification report\nprint(classification_report(y_train, xgbc1.predict(x_train)))","metadata":{"execution":{"iopub.status.busy":"2022-07-18T14:41:16.551682Z","iopub.execute_input":"2022-07-18T14:41:16.553063Z","iopub.status.idle":"2022-07-18T14:41:16.607936Z","shell.execute_reply.started":"2022-07-18T14:41:16.553011Z","shell.execute_reply":"2022-07-18T14:41:16.607115Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Performance metrics\nxgbc1_perf_results = xgbc1.evals_result()","metadata":{"execution":{"iopub.status.busy":"2022-07-18T14:41:19.814218Z","iopub.execute_input":"2022-07-18T14:41:19.815012Z","iopub.status.idle":"2022-07-18T14:41:19.819681Z","shell.execute_reply.started":"2022-07-18T14:41:19.814973Z","shell.execute_reply":"2022-07-18T14:41:19.818670Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plot learning curves\nplt.figure(figsize=(14,14))\nplt.plot(xgbc1_perf_results['validation_0']['logloss'], label='train')\nplt.plot(xgbc1_perf_results['validation_1']['logloss'], label='validation')\n\n# Show where performance flattens out\n#plt.axvline(x = 350, color='black')\n\n# Labels\nplt.suptitle('Log Loss for Training and Validation Sets')\nplt.xlabel('n_estimators')\nplt.ylabel('LogLoss')\n# show the legend\nplt.legend()\n# show the plot\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-18T14:41:21.668753Z","iopub.execute_input":"2022-07-18T14:41:21.669550Z","iopub.status.idle":"2022-07-18T14:41:21.980033Z","shell.execute_reply.started":"2022-07-18T14:41:21.669512Z","shell.execute_reply":"2022-07-18T14:41:21.978889Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# plot classification error\nplt.figure(figsize=(14,14))\nplt.plot(xgbc1_perf_results['validation_0']['error'], label='train')\nplt.plot(xgbc1_perf_results['validation_1']['error'], label='validation')\nplt.legend()\nplt.ylabel('Classification Error')\nplt.title('XGBoost Classification Error')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-18T14:41:26.279764Z","iopub.execute_input":"2022-07-18T14:41:26.280574Z","iopub.status.idle":"2022-07-18T14:41:26.551904Z","shell.execute_reply.started":"2022-07-18T14:41:26.280534Z","shell.execute_reply":"2022-07-18T14:41:26.550888Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Predictions\nxgbc1_preds = xgbc1.predict(test)","metadata":{"execution":{"iopub.status.busy":"2022-07-18T14:43:28.790915Z","iopub.execute_input":"2022-07-18T14:43:28.791447Z","iopub.status.idle":"2022-07-18T14:43:28.857331Z","shell.execute_reply.started":"2022-07-18T14:43:28.791400Z","shell.execute_reply":"2022-07-18T14:43:28.856056Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create a new dataframe\ndata={'PassengerId':passenger_id, 'Transported':xgbc1_preds}\n\nxgbc1_preds_df=pd.DataFrame(data=data)\n\n# Check\nxgbc1_preds_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-18T14:44:24.426987Z","iopub.execute_input":"2022-07-18T14:44:24.427370Z","iopub.status.idle":"2022-07-18T14:44:24.441555Z","shell.execute_reply.started":"2022-07-18T14:44:24.427339Z","shell.execute_reply":"2022-07-18T14:44:24.440178Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Replacements\nxgbc1_preds_df.replace(to_replace=0, value='False', inplace=True)\nxgbc1_preds_df.replace(to_replace=1, value='True', inplace=True)\n\n# Check\nxgbc1_preds_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-18T14:45:41.601591Z","iopub.execute_input":"2022-07-18T14:45:41.602144Z","iopub.status.idle":"2022-07-18T14:45:41.620464Z","shell.execute_reply.started":"2022-07-18T14:45:41.602101Z","shell.execute_reply":"2022-07-18T14:45:41.619335Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create the CSV for submission\nxgbc1_preds_df.to_csv('xgbc1_submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-18T14:45:53.383018Z","iopub.execute_input":"2022-07-18T14:45:53.383472Z","iopub.status.idle":"2022-07-18T14:45:53.401956Z","shell.execute_reply.started":"2022-07-18T14:45:53.383436Z","shell.execute_reply":"2022-07-18T14:45:53.400613Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Random Forest + GridSearchCV","metadata":{}},{"cell_type":"code","source":"# Imports\nfrom sklearn.ensemble import RandomForestClassifier","metadata":{"execution":{"iopub.status.busy":"2022-07-18T14:55:47.966284Z","iopub.execute_input":"2022-07-18T14:55:47.966820Z","iopub.status.idle":"2022-07-18T14:55:48.066611Z","shell.execute_reply.started":"2022-07-18T14:55:47.966771Z","shell.execute_reply":"2022-07-18T14:55:48.065343Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Initialize a RandomForestClassifier object\nrfc = RandomForestClassifier(random_state=42)","metadata":{"execution":{"iopub.status.busy":"2022-07-18T14:55:49.855509Z","iopub.execute_input":"2022-07-18T14:55:49.855922Z","iopub.status.idle":"2022-07-18T14:55:49.862122Z","shell.execute_reply.started":"2022-07-18T14:55:49.855891Z","shell.execute_reply":"2022-07-18T14:55:49.860440Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Params list\nrfc_params = {'n_estimators': [200, 300, 500, 1000],\n              'max_features': ['auto', 'log2'],\n              'max_depth' : [6, 8, 10],\n              #'min_impurity_decrease': [0, 0.1],\n              #'min_samples_leaf': [3, 4, 5],\n              #'min_samples_split': [8, 10, 12],\n              'criterion' :['gini', 'entropy']}","metadata":{"execution":{"iopub.status.busy":"2022-07-18T14:56:34.920067Z","iopub.execute_input":"2022-07-18T14:56:34.921235Z","iopub.status.idle":"2022-07-18T14:56:34.927765Z","shell.execute_reply.started":"2022-07-18T14:56:34.921185Z","shell.execute_reply":"2022-07-18T14:56:34.926373Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# New grid search object\ngs_rfc = GridSearchCV(rfc, param_grid=rfc_params, cv=10, n_jobs=-1, return_train_score=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-18T14:56:36.445490Z","iopub.execute_input":"2022-07-18T14:56:36.445970Z","iopub.status.idle":"2022-07-18T14:56:36.451842Z","shell.execute_reply.started":"2022-07-18T14:56:36.445933Z","shell.execute_reply":"2022-07-18T14:56:36.450487Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Fit the grid search object\ngs_rfc.fit(x_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2022-07-18T14:56:39.100578Z","iopub.execute_input":"2022-07-18T14:56:39.101862Z","iopub.status.idle":"2022-07-18T15:07:20.407922Z","shell.execute_reply.started":"2022-07-18T14:56:39.101803Z","shell.execute_reply":"2022-07-18T15:07:20.406323Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Estimator list\n# To a dataframe\ngs_rfc_results_pd = pd.DataFrame(gs_rfc.cv_results_)\n\n# View\ngs_rfc_results_pd","metadata":{"execution":{"iopub.status.busy":"2022-07-18T15:10:51.874582Z","iopub.execute_input":"2022-07-18T15:10:51.875115Z","iopub.status.idle":"2022-07-18T15:10:51.959235Z","shell.execute_reply.started":"2022-07-18T15:10:51.875064Z","shell.execute_reply":"2022-07-18T15:10:51.958056Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Best parameters\ngs_rfc.best_params_","metadata":{"execution":{"iopub.status.busy":"2022-07-18T15:10:59.543042Z","iopub.execute_input":"2022-07-18T15:10:59.543472Z","iopub.status.idle":"2022-07-18T15:10:59.552196Z","shell.execute_reply.started":"2022-07-18T15:10:59.543441Z","shell.execute_reply":"2022-07-18T15:10:59.550604Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Random Forest Classifier from Best Parameters","metadata":{}},{"cell_type":"code","source":"# New RandomForestClassifier object\nrfc1 = RandomForestClassifier(max_depth=10,\n                              max_features='auto',\n                              n_estimators=1000, \n                              #min_samples_leaf=3,\n                              #min_samples_split=8,\n                              criterion='gini',\n                              random_state=42)","metadata":{"execution":{"iopub.status.busy":"2022-07-18T15:11:15.459406Z","iopub.execute_input":"2022-07-18T15:11:15.459843Z","iopub.status.idle":"2022-07-18T15:11:15.465726Z","shell.execute_reply.started":"2022-07-18T15:11:15.459806Z","shell.execute_reply":"2022-07-18T15:11:15.464594Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Fit\nrfc1.fit(x_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2022-07-18T15:11:19.812282Z","iopub.execute_input":"2022-07-18T15:11:19.812665Z","iopub.status.idle":"2022-07-18T15:11:26.698186Z","shell.execute_reply.started":"2022-07-18T15:11:19.812635Z","shell.execute_reply":"2022-07-18T15:11:26.696803Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Training Performance","metadata":{}},{"cell_type":"code","source":"# Accuracy score\nprint(f\"Training accuracy: {accuracy_score(y_train, rfc1.predict(x_train))}.\")","metadata":{"execution":{"iopub.status.busy":"2022-07-18T15:11:26.700137Z","iopub.execute_input":"2022-07-18T15:11:26.700489Z","iopub.status.idle":"2022-07-18T15:11:27.944517Z","shell.execute_reply.started":"2022-07-18T15:11:26.700459Z","shell.execute_reply":"2022-07-18T15:11:27.943212Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Classification report\nprint(classification_report(y_train, rfc1.predict(x_train)))","metadata":{"execution":{"iopub.status.busy":"2022-07-18T15:11:32.420140Z","iopub.execute_input":"2022-07-18T15:11:32.421104Z","iopub.status.idle":"2022-07-18T15:11:33.667863Z","shell.execute_reply.started":"2022-07-18T15:11:32.421059Z","shell.execute_reply":"2022-07-18T15:11:33.666942Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Confusion Matrix- training set\nplot_confusion_matrix(rfc1, x_train, y_train, cmap='rocket')","metadata":{"execution":{"iopub.status.busy":"2022-07-18T15:11:35.271231Z","iopub.execute_input":"2022-07-18T15:11:35.271657Z","iopub.status.idle":"2022-07-18T15:11:36.734546Z","shell.execute_reply.started":"2022-07-18T15:11:35.271622Z","shell.execute_reply":"2022-07-18T15:11:36.733132Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Validation Performance","metadata":{}},{"cell_type":"code","source":"# Accuracy score\nprint(f\"Validation accuracy: {accuracy_score(y_validation, rfc1.predict(x_validation))}.\")","metadata":{"execution":{"iopub.status.busy":"2022-07-18T15:11:44.272907Z","iopub.execute_input":"2022-07-18T15:11:44.273340Z","iopub.status.idle":"2022-07-18T15:11:44.571026Z","shell.execute_reply.started":"2022-07-18T15:11:44.273307Z","shell.execute_reply":"2022-07-18T15:11:44.569834Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Validation\nprint(classification_report(y_validation, rfc1.predict(x_validation)))","metadata":{"execution":{"iopub.status.busy":"2022-07-18T15:11:48.360148Z","iopub.execute_input":"2022-07-18T15:11:48.360533Z","iopub.status.idle":"2022-07-18T15:11:48.660232Z","shell.execute_reply.started":"2022-07-18T15:11:48.360504Z","shell.execute_reply":"2022-07-18T15:11:48.659034Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Confusion matrix\nplot_confusion_matrix(rfc1, x_validation, y_validation, cmap='rocket')","metadata":{"execution":{"iopub.status.busy":"2022-07-18T15:11:55.830142Z","iopub.execute_input":"2022-07-18T15:11:55.830566Z","iopub.status.idle":"2022-07-18T15:11:56.338227Z","shell.execute_reply.started":"2022-07-18T15:11:55.830531Z","shell.execute_reply":"2022-07-18T15:11:56.337350Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Predictions on test set\nrfc_predictions = rfc1.predict(test)","metadata":{"execution":{"iopub.status.busy":"2022-07-18T15:12:10.925995Z","iopub.execute_input":"2022-07-18T15:12:10.926420Z","iopub.status.idle":"2022-07-18T15:12:11.705069Z","shell.execute_reply.started":"2022-07-18T15:12:10.926387Z","shell.execute_reply":"2022-07-18T15:12:11.703787Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create a data frame\ndata = {'PassengerId': passenger_id,\n        'Transported': rfc_predictions}\n\nrfc_pred_df=pd.DataFrame(data=data)\n\n# Check\nrfc_pred_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-18T15:12:15.034723Z","iopub.execute_input":"2022-07-18T15:12:15.035166Z","iopub.status.idle":"2022-07-18T15:12:15.049061Z","shell.execute_reply.started":"2022-07-18T15:12:15.035131Z","shell.execute_reply":"2022-07-18T15:12:15.047517Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Change 1 -> True, 0 -> False\nrfc_pred_df.replace(to_replace=0, value='False', inplace=True)\nrfc_pred_df.replace(to_replace=1, value='True', inplace=True)\n\n# Check\nrfc_pred_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-18T15:12:17.639607Z","iopub.execute_input":"2022-07-18T15:12:17.640951Z","iopub.status.idle":"2022-07-18T15:12:17.656544Z","shell.execute_reply.started":"2022-07-18T15:12:17.640907Z","shell.execute_reply":"2022-07-18T15:12:17.655261Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Export to CSV\nrfc_pred_df.to_csv('rfc_preds.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-18T00:54:10.371479Z","iopub.execute_input":"2022-07-18T00:54:10.371879Z","iopub.status.idle":"2022-07-18T00:54:10.386457Z","shell.execute_reply.started":"2022-07-18T00:54:10.371847Z","shell.execute_reply":"2022-07-18T00:54:10.384918Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Condensed Cost Model","metadata":{}},{"cell_type":"code","source":"# New training set\ntrain2 = pd.read_csv('../input/spaceship-titanic/train.csv')\n\n# Check\ntrain2.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-18T15:13:10.277756Z","iopub.execute_input":"2022-07-18T15:13:10.278201Z","iopub.status.idle":"2022-07-18T15:13:10.333010Z","shell.execute_reply.started":"2022-07-18T15:13:10.278168Z","shell.execute_reply":"2022-07-18T15:13:10.331749Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Check dtypes\ntrain2.dtypes","metadata":{"execution":{"iopub.status.busy":"2022-07-18T15:13:12.163017Z","iopub.execute_input":"2022-07-18T15:13:12.163421Z","iopub.status.idle":"2022-07-18T15:13:12.172466Z","shell.execute_reply.started":"2022-07-18T15:13:12.163389Z","shell.execute_reply":"2022-07-18T15:13:12.171540Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Handling missing values\n# Columns with missing values\ntrain2_nan_cols = train2.columns[train2.isnull().any()]\n\n# Replace missing values in Age with mean\ntrain2.Age.fillna(train2.Age.mean(), inplace=True)\n\n# Replace missing values in RoomService, FoodCourt, ShoppingMall, Spa, and VRDeck with median\n# Replace categorical values with most common values\nfor col in train2_nan_cols:\n    if train2[col].dtype == 'object':\n        train2[col].fillna(train2[col].mode().iloc[0], inplace=True)\n    elif train2[col].dtype == 'float64':\n        if train2[col].name != 'Age':\n            train2[col].fillna(train2[col].median(), inplace=True)\n        else:\n            # Do nothing\n            continue","metadata":{"execution":{"iopub.status.busy":"2022-07-18T15:13:13.150322Z","iopub.execute_input":"2022-07-18T15:13:13.153005Z","iopub.status.idle":"2022-07-18T15:13:13.199435Z","shell.execute_reply.started":"2022-07-18T15:13:13.152961Z","shell.execute_reply":"2022-07-18T15:13:13.198393Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Check there are no nulls\ntrain2.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2022-07-18T15:13:25.978550Z","iopub.execute_input":"2022-07-18T15:13:25.978995Z","iopub.status.idle":"2022-07-18T15:13:25.998694Z","shell.execute_reply.started":"2022-07-18T15:13:25.978958Z","shell.execute_reply":"2022-07-18T15:13:25.997298Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Drop Name from train and test\ntrain2.drop(['Name'], axis = 1, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-18T15:13:27.199901Z","iopub.execute_input":"2022-07-18T15:13:27.200283Z","iopub.status.idle":"2022-07-18T15:13:27.207873Z","shell.execute_reply.started":"2022-07-18T15:13:27.200253Z","shell.execute_reply":"2022-07-18T15:13:27.206925Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create a new variable that sums up the cost across RoomService, FoodCourt, ShoppingMall, Spa, and VRDeck\n# Hold the value per row\ntrain_total_spend =[]\n\n# Iterate through each column in the list\nfor i in range(len(train2.RoomService)):\n    spend_sum = train2.RoomService[i] + train2.FoodCourt[i] + train2.ShoppingMall[i] + train2.Spa[i] + train2.VRDeck[i]\n    train_total_spend.append(spend_sum)\n\n# Check\nlen(train_total_spend)","metadata":{"execution":{"iopub.status.busy":"2022-07-18T15:13:29.660036Z","iopub.execute_input":"2022-07-18T15:13:29.660626Z","iopub.status.idle":"2022-07-18T15:13:30.208738Z","shell.execute_reply.started":"2022-07-18T15:13:29.660593Z","shell.execute_reply":"2022-07-18T15:13:30.207567Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Insert\ntrain2.insert(12,'tot_spend', train_total_spend)\n\n# Log (x+1) transformation\ntrain2['tot_spend'] = logTx(train2.tot_spend)\n\n# Check\ntrain2.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-18T15:13:33.824961Z","iopub.execute_input":"2022-07-18T15:13:33.825710Z","iopub.status.idle":"2022-07-18T15:13:33.856911Z","shell.execute_reply.started":"2022-07-18T15:13:33.825671Z","shell.execute_reply":"2022-07-18T15:13:33.855565Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Get companion count for PassengerId in train data set\ntrain2_companion_count = passengerIdSplitter(train2.PassengerId)","metadata":{"execution":{"iopub.status.busy":"2022-07-18T15:13:35.326077Z","iopub.execute_input":"2022-07-18T15:13:35.326465Z","iopub.status.idle":"2022-07-18T15:13:37.880372Z","shell.execute_reply.started":"2022-07-18T15:13:35.326436Z","shell.execute_reply":"2022-07-18T15:13:37.879101Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Add\ntrain2.insert(1, 'companion_count', train2_companion_count)\n\n# Check\ntrain2.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-18T15:13:37.882230Z","iopub.execute_input":"2022-07-18T15:13:37.882566Z","iopub.status.idle":"2022-07-18T15:13:37.910225Z","shell.execute_reply.started":"2022-07-18T15:13:37.882538Z","shell.execute_reply":"2022-07-18T15:13:37.909130Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Get deck, level, and side for train dataframe\ntrain2_deck, train2_level, train2_side = cabinSplitter(train2.Cabin)","metadata":{"execution":{"iopub.status.busy":"2022-07-18T15:13:41.366434Z","iopub.execute_input":"2022-07-18T15:13:41.366922Z","iopub.status.idle":"2022-07-18T15:13:41.382418Z","shell.execute_reply.started":"2022-07-18T15:13:41.366878Z","shell.execute_reply":"2022-07-18T15:13:41.381217Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Add to the training data set\ntrain2.insert(4, 'deck', train2_deck)\ntrain2.insert(5, 'level', train2_level)\ntrain2.insert(6, 'side', train2_side)\n\n# Check\ntrain2.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-18T15:13:58.730362Z","iopub.execute_input":"2022-07-18T15:13:58.730797Z","iopub.status.idle":"2022-07-18T15:13:58.765318Z","shell.execute_reply.started":"2022-07-18T15:13:58.730750Z","shell.execute_reply":"2022-07-18T15:13:58.763947Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Get the categories\ntrain2_level_cat = levelCategorizer(train2.level)","metadata":{"execution":{"iopub.status.busy":"2022-07-18T15:14:01.539709Z","iopub.execute_input":"2022-07-18T15:14:01.540145Z","iopub.status.idle":"2022-07-18T15:14:01.551703Z","shell.execute_reply.started":"2022-07-18T15:14:01.540110Z","shell.execute_reply":"2022-07-18T15:14:01.550499Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Insert level_range\ntrain2.insert(5, 'level_range', train2_level_cat)","metadata":{"execution":{"iopub.status.busy":"2022-07-18T15:14:02.795669Z","iopub.execute_input":"2022-07-18T15:14:02.796367Z","iopub.status.idle":"2022-07-18T15:14:02.802418Z","shell.execute_reply.started":"2022-07-18T15:14:02.796317Z","shell.execute_reply":"2022-07-18T15:14:02.801162Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Drop PassengerId, Cabin, level, RoomService, FoodCourt, ShoppingMall, Spa, VRDeck\ntrain2.drop(['PassengerId', 'Cabin','level', 'RoomService', 'FoodCourt', 'ShoppingMall', 'Spa', 'VRDeck'], axis=1, inplace=True)\n\n# Check\ntrain2.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-18T15:14:06.682543Z","iopub.execute_input":"2022-07-18T15:14:06.682954Z","iopub.status.idle":"2022-07-18T15:14:06.706701Z","shell.execute_reply.started":"2022-07-18T15:14:06.682920Z","shell.execute_reply":"2022-07-18T15:14:06.705923Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Initialize OrdinalEncoder instance\nord2 = OrdinalEncoder(dtype = 'int64')","metadata":{"execution":{"iopub.status.busy":"2022-07-18T15:14:11.679821Z","iopub.execute_input":"2022-07-18T15:14:11.680271Z","iopub.status.idle":"2022-07-18T15:14:11.685281Z","shell.execute_reply.started":"2022-07-18T15:14:11.680234Z","shell.execute_reply":"2022-07-18T15:14:11.684300Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Encoding variables\ntrain2.HomePlanet = ord2.fit_transform(train2.HomePlanet.to_numpy().reshape(-1,1))\ntrain2.CryoSleep = ord2.fit_transform(train2.CryoSleep.to_numpy().reshape(-1,1))\ntrain2.deck = ord2.fit_transform(train2.deck.to_numpy().reshape(-1,1))\ntrain2.level_range = ord2.fit_transform(train2.level_range.to_numpy().reshape(-1,1))\ntrain2.side = ord2.fit_transform(train2.side.to_numpy().reshape(-1,1))\ntrain2.Destination = ord2.fit_transform(train2.Destination.to_numpy().reshape(-1,1))\ntrain2.VIP = ord2.fit_transform(train2.VIP.to_numpy().reshape(-1,1))\n\n# Check\ntrain2.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-18T15:14:13.626534Z","iopub.execute_input":"2022-07-18T15:14:13.626966Z","iopub.status.idle":"2022-07-18T15:14:13.669927Z","shell.execute_reply.started":"2022-07-18T15:14:13.626917Z","shell.execute_reply":"2022-07-18T15:14:13.668885Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# LabelEncoder\nenc2 = LabelEncoder()\n\n# Fit transform\ntrain2['Transported'] = enc2.fit_transform(train2.Transported)\n\n# Check\ntrain2.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-18T15:14:16.711167Z","iopub.execute_input":"2022-07-18T15:14:16.712013Z","iopub.status.idle":"2022-07-18T15:14:16.727897Z","shell.execute_reply.started":"2022-07-18T15:14:16.711975Z","shell.execute_reply":"2022-07-18T15:14:16.726910Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Visualizations","metadata":{}},{"cell_type":"code","source":"# Heatmap\nplt.figure(figsize=(20,20))\nsns.heatmap(train2.corr(), annot=True, fmt='g')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-18T15:14:27.014763Z","iopub.execute_input":"2022-07-18T15:14:27.015179Z","iopub.status.idle":"2022-07-18T15:14:27.955999Z","shell.execute_reply.started":"2022-07-18T15:14:27.015148Z","shell.execute_reply":"2022-07-18T15:14:27.954921Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"There are several variables that have low predictive value regarding the target variable, Transported:\n- companion_count, 0.08\n- level_range, 0.08\n- Age, -0.07\n- VIP, -0.04\n\ntot_spend also appears to exhibit high multicollinearity with CryoSleep, which is a one-hot encoded predictor variable that would be expected to have a relationship with tot_spend:\n- tot_spend would be higher for those not in CryoSleep as they might need to spend money on food and entertainment.\n\nVIF values will be checked to determine if the multicollinearity is excessive to the point where it would impact the model.\n$$VIF = \\frac{1}{1-R^{2}}$$","metadata":{}},{"cell_type":"code","source":"# Check VIF\ncalcVIF(train2)","metadata":{"execution":{"iopub.status.busy":"2022-07-18T15:14:52.960826Z","iopub.execute_input":"2022-07-18T15:14:52.961266Z","iopub.status.idle":"2022-07-18T15:14:53.079665Z","shell.execute_reply.started":"2022-07-18T15:14:52.961235Z","shell.execute_reply":"2022-07-18T15:14:53.078282Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Following the rule that a VIF >= 10 is potentially troublesome for a model, it would appear that all of the independent variables except those with very low $R^{2}$ values in the model are safe to retain. However, to start, all variables will be retained regardless of their low predictive value.","metadata":{}},{"cell_type":"code","source":"# Remove companion_count, level_range, Age, VIP\ntrain2.drop(['Transported'], axis=1, inplace=True)\n\n# Check\ntrain2.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-18T15:15:42.422105Z","iopub.execute_input":"2022-07-18T15:15:42.422534Z","iopub.status.idle":"2022-07-18T15:15:42.440211Z","shell.execute_reply.started":"2022-07-18T15:15:42.422501Z","shell.execute_reply":"2022-07-18T15:15:42.439279Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Split data\nx_train2, x_validation2, y_train2, y_validation2 = train_test_split(train2, y, test_size=0.1, random_state=42)","metadata":{"execution":{"iopub.status.busy":"2022-07-18T15:15:56.863987Z","iopub.execute_input":"2022-07-18T15:15:56.864444Z","iopub.status.idle":"2022-07-18T15:15:56.873766Z","shell.execute_reply.started":"2022-07-18T15:15:56.864408Z","shell.execute_reply":"2022-07-18T15:15:56.872625Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Evaluation sets for each iteration\neval_sets2 = [(x_train2, y_train2), (x_validation2, y_validation2)]","metadata":{"execution":{"iopub.status.busy":"2022-07-18T15:15:58.956745Z","iopub.execute_input":"2022-07-18T15:15:58.957770Z","iopub.status.idle":"2022-07-18T15:15:58.962764Z","shell.execute_reply.started":"2022-07-18T15:15:58.957716Z","shell.execute_reply":"2022-07-18T15:15:58.961809Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# New model\n# Initialize a new XGBClassifier instance\nxgbc2 = XGBClassifier(objective='binary:logistic',\n                      nthread=4,\n                      seed=42,\n                      eval_metric = ['error','logloss']) ","metadata":{"execution":{"iopub.status.busy":"2022-07-18T15:16:04.807073Z","iopub.execute_input":"2022-07-18T15:16:04.807499Z","iopub.status.idle":"2022-07-18T15:16:04.813432Z","shell.execute_reply.started":"2022-07-18T15:16:04.807466Z","shell.execute_reply":"2022-07-18T15:16:04.812132Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Parameter list - start with the most important parameters\nnew_params={'max_depth': [3, 4, 5, 6],\n            'n_estimators': [100, 150, 300],\n            'eta': [0.05, 0.1, 0.03]}","metadata":{"execution":{"iopub.status.busy":"2022-07-18T15:17:14.357221Z","iopub.execute_input":"2022-07-18T15:17:14.357595Z","iopub.status.idle":"2022-07-18T15:17:14.363679Z","shell.execute_reply.started":"2022-07-18T15:17:14.357564Z","shell.execute_reply":"2022-07-18T15:17:14.362816Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# New grid search object\ngc2 = GridSearchCV(xgbc2, param_grid=new_params, n_jobs=-1, cv=10)","metadata":{"execution":{"iopub.status.busy":"2022-07-18T15:17:16.317212Z","iopub.execute_input":"2022-07-18T15:17:16.318205Z","iopub.status.idle":"2022-07-18T15:17:16.322501Z","shell.execute_reply.started":"2022-07-18T15:17:16.318162Z","shell.execute_reply":"2022-07-18T15:17:16.321668Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Fit the grid search object\ngc2.fit(x_train2, y_train2)","metadata":{"execution":{"iopub.status.busy":"2022-07-18T15:17:17.403994Z","iopub.execute_input":"2022-07-18T15:17:17.404922Z","iopub.status.idle":"2022-07-18T18:05:50.097208Z","shell.execute_reply.started":"2022-07-18T15:17:17.404885Z","shell.execute_reply":"2022-07-18T18:05:50.095007Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# To a dataframe\ngc2_results_pd = pd.DataFrame(gc2.cv_results_)\n\n# View\ngc2_results_pd","metadata":{"execution":{"iopub.status.busy":"2022-07-18T18:32:27.349986Z","iopub.execute_input":"2022-07-18T18:32:27.351929Z","iopub.status.idle":"2022-07-18T18:32:27.425226Z","shell.execute_reply.started":"2022-07-18T18:32:27.351876Z","shell.execute_reply":"2022-07-18T18:32:27.424254Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# View the best parameters\ngc2.best_params_","metadata":{"execution":{"iopub.status.busy":"2022-07-18T18:32:36.228521Z","iopub.execute_input":"2022-07-18T18:32:36.228928Z","iopub.status.idle":"2022-07-18T18:32:36.236121Z","shell.execute_reply.started":"2022-07-18T18:32:36.228896Z","shell.execute_reply":"2022-07-18T18:32:36.234930Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## New Model with Best Parameters","metadata":{}},{"cell_type":"code","source":"model = XGBClassifier(objective='binary:logistic',\n                      eta=0.03,\n                      max_depth=4,\n                      #max_leaves=30,\n                      n_estimators=300,\n                      subsample=0.9,\n                      gamma = 1.5,\n                      nthread=4,\n                      seed=42,\n                      eval_metric = ['error','logloss'])","metadata":{"execution":{"iopub.status.busy":"2022-07-18T18:41:23.758330Z","iopub.execute_input":"2022-07-18T18:41:23.758890Z","iopub.status.idle":"2022-07-18T18:41:23.767201Z","shell.execute_reply.started":"2022-07-18T18:41:23.758821Z","shell.execute_reply":"2022-07-18T18:41:23.766165Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Fit\nmodel.fit(x_train2, y_train2, eval_set=eval_sets2)","metadata":{"execution":{"iopub.status.busy":"2022-07-18T18:41:24.943755Z","iopub.execute_input":"2022-07-18T18:41:24.944674Z","iopub.status.idle":"2022-07-18T18:41:28.272478Z","shell.execute_reply.started":"2022-07-18T18:41:24.944636Z","shell.execute_reply":"2022-07-18T18:41:28.271533Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Accuracy score\nmodel_val_acc_score = accuracy_score(y_validation2, model.predict(x_validation2))\nmodel_train_acc_score = accuracy_score(y_train2, model.predict(x_train2))\n\n# View\nprint(f\"Validation set accuracy score: {model_val_acc_score}.\")\nprint(f\"Training set accuracy score: {model_train_acc_score}.\")","metadata":{"execution":{"iopub.status.busy":"2022-07-18T18:42:35.501798Z","iopub.execute_input":"2022-07-18T18:42:35.502289Z","iopub.status.idle":"2022-07-18T18:42:35.542751Z","shell.execute_reply.started":"2022-07-18T18:42:35.502253Z","shell.execute_reply":"2022-07-18T18:42:35.541875Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Performance metrics\nmodel_perf_results = model.evals_result()","metadata":{"execution":{"iopub.status.busy":"2022-07-18T18:42:44.325462Z","iopub.execute_input":"2022-07-18T18:42:44.325975Z","iopub.status.idle":"2022-07-18T18:42:44.332521Z","shell.execute_reply.started":"2022-07-18T18:42:44.325816Z","shell.execute_reply":"2022-07-18T18:42:44.331048Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plot learning curves\nplt.figure(figsize=(14,14))\nplt.plot(model_perf_results['validation_0']['logloss'], label='train')\nplt.plot(model_perf_results['validation_1']['logloss'], label='validation')\n\n# Show where performance flattens out\n#plt.axvline(x = 350, color='black')\n\n# Labels\nplt.suptitle('Log Loss for Training and Validation Sets')\nplt.xlabel('n_estimators')\nplt.ylabel('LogLoss')\n# show the legend\nplt.legend()\n# show the plot\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-18T18:42:47.677986Z","iopub.execute_input":"2022-07-18T18:42:47.678390Z","iopub.status.idle":"2022-07-18T18:42:47.894506Z","shell.execute_reply.started":"2022-07-18T18:42:47.678358Z","shell.execute_reply":"2022-07-18T18:42:47.893274Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# plot classification error\nplt.figure(figsize=(14,14))\nplt.plot(model_perf_results['validation_0']['error'], label='train')\nplt.plot(model_perf_results['validation_1']['error'], label='validation')\nplt.legend()\nplt.ylabel('Classification Error')\nplt.title('XGBoost Classification Error')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-18T18:42:53.530894Z","iopub.execute_input":"2022-07-18T18:42:53.531292Z","iopub.status.idle":"2022-07-18T18:42:53.787887Z","shell.execute_reply.started":"2022-07-18T18:42:53.531259Z","shell.execute_reply":"2022-07-18T18:42:53.786655Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}