{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import pandas as pd\nimport time\nimport numpy as np\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\n# Import Model Libraries\nimport xgboost as xgb\nfrom sklearn import metrics\nfrom sklearn.model_selection import GridSearchCV\nfrom sklearn.metrics import accuracy_score\nfrom sklearn.model_selection import train_test_split\n\n# Train a base decision tree regressor model on the data\nfrom sklearn.tree import DecisionTreeRegressor\n\nimport warnings\nwarnings.filterwarnings('ignore')\n","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"e431b6b0a6f393e0e25777f0661e78b76d14416f"},"cell_type":"markdown","source":"# Load Boston Dataset"},{"metadata":{"trusted":true,"_uuid":"ddfe3c89ba4f28751fe2383fe759e66516a22ad9"},"cell_type":"code","source":"from sklearn.datasets import load_boston\nboston = load_boston()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"2a4ff55adc6d584d19f02f9a115930ccdba09351"},"cell_type":"code","source":"bos = pd.DataFrame(boston.data)\nbos.columns = boston.feature_names\nbos['PRICE'] = boston.target\nbos.head()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"f538bf740367b1d4015edd048e79ba130faa84d2"},"cell_type":"markdown","source":"# Train Test Split"},{"metadata":{"trusted":true,"_uuid":"d33aa2578daf7cd8b6f0a5b8789aca5493a5ed64"},"cell_type":"code","source":"# Split Train/Test Set\nX_train_2, X_test_2, y_train_2, y_test_2 = train_test_split(bos.drop([\"PRICE\"], axis=1), bos[\"PRICE\"],random_state=10, test_size=0.25)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"5656bdf7615e4cf876e2d046ae014bfa21489095"},"cell_type":"code","source":"# Examine our dataset\nX_train_2.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"82e536393cd597fc8fb1378912617e5bfd5c4dd2"},"cell_type":"code","source":"# Examine shape of the dataset\nX_train_2.shape","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"e8112eec76fad15b7b2ca69f432b0ebf0e176b10"},"cell_type":"markdown","source":"# Check for any missing values"},{"metadata":{"trusted":true,"_uuid":"77e830af1a40408213be446a9c8ec2238c0868d8"},"cell_type":"code","source":"# Check for any missing values\nX_train_2.isnull().any()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"b8efc91642355d11125a5d81070293ac081061d1"},"cell_type":"markdown","source":"# Tree Ensemble (Boosting) from Scratch"},{"metadata":{"_uuid":"14cb66d63fb903af202b3bd72a6bda327a649add"},"cell_type":"markdown","source":"**Train first tree**"},{"metadata":{"trusted":true,"_uuid":"a3495376682d67cea1690804910829d44062cb6a"},"cell_type":"code","source":"# Train a base decision tree regressor model on the data\nfrom sklearn.tree import DecisionTreeRegressor\n# Fit model\ntree_reg1 = DecisionTreeRegressor(max_depth=2)\ntree_reg1.fit(X_train_2,y_train_2)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"428ea2506a5f9cd8c12e747e0820c6510a261c17"},"cell_type":"code","source":"# Compute errors/residuals on first tree\nr1 = y_train_2 - tree_reg1.predict(X_train_2)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"c30556f9db78614d580571a8fb4af307bd9169dd"},"cell_type":"markdown","source":"**Train second tree**"},{"metadata":{"trusted":true,"_uuid":"c06c30393057939f455b41f66512001a6c93cd92"},"cell_type":"code","source":"# Fit second model\ntree_reg2 = DecisionTreeRegressor(max_depth=2)\ntree_reg2.fit(X_train_2,r1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"2363ab9e8a53fdb30d7f82085c452a5e1fb9a42f"},"cell_type":"code","source":"# Compute errors/residuals on second tree\nr2 = r1 - tree_reg2.predict(X_train_2)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"a3541a0880b2d703fb27753e5deef9f8237bf2a6"},"cell_type":"markdown","source":"**Train third tree**"},{"metadata":{"trusted":true,"_uuid":"0b70986330fe2d44fd5c6295ad6cc8ce77e0b8c6"},"cell_type":"code","source":"# Fit third model\ntree_reg3 = DecisionTreeRegressor(max_depth=2)\ntree_reg3.fit(X_train_2,r2)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"57ced2ec830df64b041e7020d1b57b8c8b969f43"},"cell_type":"markdown","source":"# ENSEMBLE: Combine all three tree predictions"},{"metadata":{"trusted":true,"_uuid":"7c40f6c10a4b3e6caf4452edfe77695fa620ade7"},"cell_type":"code","source":"# Add up the predictions of each tree model, which is our ensemble of three trees\ny_pred = sum(tree.predict(X_train_2) for tree in (tree_reg1, tree_reg2, tree_reg3))\n","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"01e634a5721717f7f311142ce0a9a5acb57097cb"},"cell_type":"markdown","source":"**First 10 ENSEMBLE Predictions**\n"},{"metadata":{"trusted":true,"_uuid":"0741d76a64f27641756e540e9f7efec57fa26f87"},"cell_type":"code","source":"y_pred[:10]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"33558b8bd7b9f39088902ab284d39bc99d1010f5"},"cell_type":"code","source":"#actual values\ny_train_2[:10]","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"861fba8be579431c0fed46498703c8ace8ac8a3e"},"cell_type":"markdown","source":"# model prediction"},{"metadata":{"trusted":true,"_uuid":"f26d29c3581e42e0fee1b9bfaefd2d1b9aee53d5"},"cell_type":"code","source":"tree_reg1.predict(X_train_2)[:10]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"54fb7b2a66f8c3c0ae9f8c276f5c7267ea760712"},"cell_type":"code","source":"# Create dataframe of all predictions\npredictions = pd.DataFrame(tree_reg1.predict(X_train_2)[:10], columns=['Model_1'])\npredictions['Model_2'] = pd.DataFrame(tree_reg2.predict(X_train_2)[:10])\npredictions['Model_3'] = pd.DataFrame(tree_reg3.predict(X_train_2)[:10])\npredictions['Ensemble'] = pd.DataFrame(y_pred[:10])\npredictions['Actual'] = y_train_2.head(10).reset_index()['PRICE']\n\n# Display predictions\npredictions","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"5de66a5f18b9c1f316829fd759c5813fcc3bb1c6"},"cell_type":"code","source":"errors = []\nfor n_estimators in [1,2,3,4,5,6,7,8,9,10]:\n    clf = xgb.XGBRegressor(max_depth=2, n_estimators=n_estimators)\n    clf.fit(X_train_2, y_train_2, verbose=False)\n    errors.append(\n        {\n            'Tree Count': n_estimators,\n            'Average Error': np.average(y_train_2 - clf.predict(X_train_2)),\n        })\n    \nn_estimators_lr = pd.DataFrame(errors).set_index('Tree Count').sort_index()\nn_estimators_lr","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"95cd76149373f69e5716dc3f49210ebcc5c043b8"},"cell_type":"markdown","source":"# Using Sklearn Gradient Boosting Regressor"},{"metadata":{"trusted":true,"_uuid":"0c1913dcff444b086b4b6e20c8bced9b54155e1e"},"cell_type":"code","source":"# Use the Sklearn GradientBoostingRegressor ensemble method to perform the same thing as the previous code above\nfrom sklearn.ensemble import GradientBoostingRegressor\n\ngbrt = GradientBoostingRegressor(max_depth=2, n_estimators=3)\ngbrt.fit(X_train_2,y_train_2)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"4b18e40bf7f08411132f0bf271cd7c9ab2b2e9fb"},"cell_type":"markdown","source":"# XGBoost Model"},{"metadata":{"trusted":true,"_uuid":"3353cde50a0128c85ed0993b4662825855451e3d"},"cell_type":"code","source":"import xgboost as xgb\nGradientBoostingRegressor()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b2dec1ee9fed8a0a534db16a38052e1800ac3138"},"cell_type":"code","source":"# Examine the default parameters\nxgb.XGBRegressor()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"447e8c13e2480e405e46c2ade64b5f8401129cb7"},"cell_type":"code","source":"# Create empty array to store results\nresults = []\n# Create watchlist to keep track of train/validation performance\neval_set = [(X_train_2, y_train_2), (X_test_2, y_test_2)]","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"6f481b13427ccb7ef80755f6f3cdce9ae3c1555f"},"cell_type":"markdown","source":"**max_depth**"},{"metadata":{"trusted":true,"_uuid":"2062cce54ac4fb30dcf41333f7dd626635cd03f3"},"cell_type":"code","source":"# Enumerate through different max_depth values and store results\nfor max_depth in [2,3,4,5,10,12,15]:\n    clf = xgb.XGBRegressor(max_depth=max_depth)\n    clf.fit(X_train_2, y_train_2, eval_set=eval_set, verbose=False)\n    results.append(\n        {\n            'max_depth': max_depth,\n            'train_error': metrics.mean_squared_error(y_train_2, clf.predict(X_train_2)),\n            'test_error': metrics.mean_squared_error(y_test_2, clf.predict(X_test_2))\n        })\n    \n# Display Results\nmax_depth_lr = pd.DataFrame(results).set_index('max_depth').sort_index()\nmax_depth_lr","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"1b95fca69ac8f3cf9dd69e1434ee6434b601d232"},"cell_type":"code","source":"# Plot Max_Depth Learning Curve\nmax_depth_lr.plot(title=\"Max_Depth Learning Curve\")","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"8048215cd5532de3579ae02c8360020386692d2e"},"cell_type":"markdown","source":"Looks like the best max_depth is 3"},{"metadata":{"_uuid":"e0b9903e6473617a883b816822ea0f3daf439ba8"},"cell_type":"markdown","source":"**Learning_Rate**"},{"metadata":{"trusted":true,"_uuid":"ad51349f351f04d8cd9149a2e4056b7035648d2a"},"cell_type":"code","source":"# Reset results array\nresults = []\n\nfor learning_rate in [0.05,0.1,0.2,0.4,0.6,0.8,1]:\n    clf = xgb.XGBRegressor(max_depth=2,learning_rate=learning_rate, n_estimators=200)\n    clf.fit(X_train_2, y_train_2, eval_set=eval_set, verbose=False)\n    results.append(\n        {\n            'learning_rate': learning_rate,\n            'train_error': metrics.mean_squared_error(y_train_2, clf.predict(X_train_2)),\n            'test_error': metrics.mean_squared_error(y_test_2, clf.predict(X_test_2))\n        })\n    \nlearning_rate_lr = pd.DataFrame(results).set_index('learning_rate').sort_index()\nlearning_rate_lr","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c7400f2b45c1a56b9a80f6ff3fd46bc2d07f0010"},"cell_type":"code","source":"# Plot Learning Rate\nlearning_rate_lr.plot(title=\"Learning Rate Learning Curve\")","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"162e8e391c72a9b29babb39b3d156954048052b8"},"cell_type":"markdown","source":"The best learning rate is 0.05"},{"metadata":{"_uuid":"8763ff9a44e3088a737a9cc4d04260a19d371d25"},"cell_type":"markdown","source":"**N_Estimators**"},{"metadata":{"trusted":true,"_uuid":"3050fe33d58e555d3856213405f203655b0e339a"},"cell_type":"code","source":"# Reset results array\nresults = []\n\nfor n_estimators in [50,60,100,150,200,500,750,1000, 1500]:\n    clf = xgb.XGBRegressor(max_depth=2,learning_rate=0.10, n_estimators=n_estimators)\n    clf.fit(X_train_2, y_train_2, eval_set=eval_set, verbose=False)\n    results.append(\n        {\n            'n_estimators': n_estimators,\n            'train_error': metrics.mean_squared_error(y_train_2, clf.predict(X_train_2)),\n            'test_error': metrics.mean_squared_error(y_test_2, clf.predict(X_test_2))\n        })\n    \nn_estimators_lr = pd.DataFrame(results).set_index('n_estimators').sort_index()\nn_estimators_lr","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"58f3e540980e39bd73007064979cf17772554fa5"},"cell_type":"code","source":"n_estimators_lr.plot(title=\"N_Estimators Learning Curve\")","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"f2adb7220bb81ee4bf3c963b8152c6f326ed3656"},"cell_type":"markdown","source":"Best N_Estimators is 50"},{"metadata":{"_uuid":"d370f255c3aed65ed708e97b626278f78176b393"},"cell_type":"markdown","source":"# GridSearchCV"},{"metadata":{"trusted":true,"_uuid":"44989ba48f07cef5916e1f15a7d91ddb45e7fdd6"},"cell_type":"code","source":"model = xgb.XGBRegressor()\n# Define Parameters\nparam_grid = {\"max_depth\": [2,3,10],\n              \"max_features\" : [1.0,0.3,0.1],\n              \"min_samples_leaf\" : [3,5,9],\n              \"n_estimators\": [50,100,300],\n              \"learning_rate\": [0.05,0.1,0.02,0.2]}\n# Perform Grid Search CV\ngs_cv = GridSearchCV(model, param_grid=param_grid, cv = 3, verbose=10, n_jobs=-1 ).fit(X_train_2, y_train_2)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c36375465c815a9bc0078502b1827280536d661f"},"cell_type":"code","source":"# Best hyperparmeter setting\ngs_cv.best_estimator_","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"825c93fb220b7e84f52b7ef4f8a34824ba03fed1"},"cell_type":"markdown","source":"**Learning Curve**"},{"metadata":{"trusted":true,"_uuid":"dbce3b64f090dd14a17fdb4204d2c324bc53263b"},"cell_type":"code","source":"# Use our best model parameters found by GridSearchCV\nbest_model = xgb.XGBRegressor(base_score=0.5, booster='gbtree', colsample_bylevel=1,\n       colsample_bytree=1, gamma=0, learning_rate=0.1, max_delta_step=0,\n       max_depth=3, max_features=1.0, min_child_weight=1,\n       min_samples_leaf=3, missing=None, n_estimators=300, n_jobs=1,\n       nthread=None, objective='reg:linear', random_state=0, reg_alpha=0,\n       reg_lambda=1, scale_pos_weight=1, seed=None, silent=True,\n       subsample=1)\n# Create eval_set\neval_set = [(X_train_2, y_train_2), (X_test_2, y_test_2)]\n\n# Fit our model to the training set\nbest_model.fit(X_train_2, y_train_2, eval_set=eval_set, verbose=False)\n\n# Make predictions with test data\ny_pred = best_model.predict(X_test_2)\npredictions = [round(value) for value in y_pred]\n\n# Retrieve performance metrics\nresults = best_model.evals_result()\nepochs = len(results['validation_0']['rmse'])\nx_axis = range(0, epochs)\n\n# Plot log loss curve\nfig, ax = plt.subplots()\nax.plot(x_axis, results['validation_0']['rmse'], label='Train')\nax.plot(x_axis, results['validation_1']['rmse'], label='Test')\nax.legend()\nplt.ylabel('RMSE')\nplt.title('XGBoost RMSE')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ed68f8834a4ddc44ecb9c8ee61d13aa1befbb421"},"cell_type":"code","source":"# Fit the training set and apply early stopping \nbest_model.fit(X_train_2, y_train_2, early_stopping_rounds=10, eval_set=eval_set, verbose=True)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"f25346ccce6a89f17586280d3a3f3023a1b7458c"},"cell_type":"markdown","source":"# Feature Importance"},{"metadata":{"trusted":true,"_uuid":"3dbf74f8d2f9679644173f63adea22588a6211a5"},"cell_type":"code","source":"# Plot basic feature importance chart\nfig, ax = plt.subplots(figsize=(12,12))\nxgb.plot_importance(best_model, height=0.5, ax=ax)\nplt.show()","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}