{"cells":[{"metadata":{"_uuid":"2d3ec42ffa8db7c12130da0183f3eb52af2af55c"},"cell_type":"markdown","source":""},{"metadata":{"_uuid":"3cf026030e76038b11b0df3aa033d0ae633a083e"},"cell_type":"markdown","source":"# Introduction\n\nThis is a simple solution applying Cross Validation through **RandomizedSearchCV** and using numeric columns."},{"metadata":{"_uuid":"00f569c81c4d9644cc7683687231dd53dea4f992"},"cell_type":"markdown","source":"## Read the train data\n"},{"metadata":{"_uuid":"dabff602a057a6ca4e8f91b8ce7dc0ddd01f6c5d","trusted":true},"cell_type":"code","source":"# Code you have previously used to load data\nimport pandas as pd\nfrom sklearn.ensemble import RandomForestRegressor\nfrom sklearn.metrics import mean_absolute_error\nfrom sklearn.model_selection import train_test_split\nfrom learntools.core import *\n\nfrom sklearn.model_selection import RandomizedSearchCV\nimport numpy as np\n\nfrom sklearn.impute import SimpleImputer\n\n\n\n# Path of the file to read. We changed the directory structure to simplify submitting to a competition\niowa_file_path = '../input/train.csv'\nhome_data = pd.read_csv(iowa_file_path)\n\n","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"bb02656062d2bb1848696874cb13c59826f16c7c"},"cell_type":"markdown","source":"# Creating a Model For the Competition\n\nBuild a Random Forest model and train it on all of **X** and **y**.  "},{"metadata":{"_uuid":"fcc5319906dcf753da3342d689f934b55270883e"},"cell_type":"markdown","source":"**1: Select Columns and Clean Data**\n\n"},{"metadata":{"trusted":true,"_uuid":"39b550995ff4721f32679480ddad4a51ccbec8ae"},"cell_type":"code","source":"#Option 1: Select certain columns \n\n#features = ['LotArea', 'YearBuilt', '1stFlrSF', '2ndFlrSF', 'FullBath', 'BedroomAbvGr', 'TotRmsAbvGrd']\n#X = home_data[features]\n\n# Split into validation and training data\n#train_X, val_X, train_y, val_y = train_test_split(X, y, random_state=1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"0640a499b2c977118fa88b6ea240f241a42bef21"},"cell_type":"code","source":"\n#Option 2: Select only numeric columns without nulls \n\ndef score_dataset(X_train, X_test, y_train, y_test):\n    model = RandomForestRegressor()\n    model.fit(X_train, y_train)\n    preds = model.predict(X_test)\n    return mean_absolute_error(y_test, preds)\n\n\n# Load data\nhome_data_target = home_data.SalePrice\nhome_data_predictors = home_data.drop(['SalePrice','Id'], axis=1)\n# For the sake of keeping the example simple, we'll use only numeric predictors.\nhome_data_numeric_predictors = home_data_predictors.select_dtypes(exclude=['object'])\nX_train, X_test, y_train, y_test = train_test_split(home_data_numeric_predictors, home_data_target,\n                                                    train_size=0.7, test_size=0.3, random_state=42)\ncols_with_missing = [col for col in X_train.columns if X_train[col].isnull().any()]\n\n\n\n\n# Method 1  **************************\nreduced_X_train = X_train.drop(cols_with_missing, axis=1)\nreduced_X_test = X_test.drop(cols_with_missing, axis=1)\nprint(\"Mean Absolute Error from dropping columns with Missing Values:\")\nprint(score_dataset(reduced_X_train, reduced_X_test, y_train, y_test))\n\n# Method 2 ************************\nmy_imputer = SimpleImputer()\nimputed_X_train = my_imputer.fit_transform(X_train)\nimputed_X_test = my_imputer.transform(X_test)\nprint(\"Mean Absolute Error from Imputation:\")\nprint(score_dataset(imputed_X_train, imputed_X_test, y_train, y_test))\n\nprint('\\n\\n');\n\n# Method 3 **************************\nimputed_X_train_plus = X_train.copy()\nimputed_X_test_plus = X_test.copy()\ncols_with_missing = (col for col in X_train.columns if X_train[col].isnull().any())\nfor col in cols_with_missing:\n    imputed_X_train_plus[col + '_was_missing'] = imputed_X_train_plus[col].isnull()\n    imputed_X_test_plus[col + '_was_missing'] = imputed_X_test_plus[col].isnull()\nmy_imputer = SimpleImputer()\nimputed_X_train_plus = my_imputer.fit_transform(imputed_X_train_plus)\nimputed_X_test_plus = my_imputer.transform(imputed_X_test_plus)\nprint(\"Mean Absolute Error from Imputation while Track What Was Imputed:\")\nprint(score_dataset(imputed_X_train_plus, imputed_X_test_plus, y_train, y_test))\nprint('\\n');\n","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"991fb051d0d0b17b190207c04353bea78858b3af"},"cell_type":"markdown","source":"**1.1: Select Clean Option and Method**\n"},{"metadata":{"trusted":true,"_uuid":"791996fa5774cbf9bfcd8e49910a77d9e5828abc"},"cell_type":"code","source":"    \ntrain_X = reduced_X_train\nval_X = reduced_X_test\n\ntrain_y =  y_train\nval_y =  y_test","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"e3ebbd30f02e2daade5145fc107113a296465795"},"cell_type":"markdown","source":"**1.2: Display Train Head AND Selected Columns**\n "},{"metadata":{"trusted":true,"_uuid":"ac6c75c98b67df0bf6425440b46a69568124133c"},"cell_type":"code","source":"\n##Head\nprint(train_X.head())\n\n##Numeric columns list\nprint(list(reduced_X_train.columns.values))\nprint('\\n\\n\\n');\n","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"f556ef9032e3311b132067a85d93101bddf10ca5"},"cell_type":"markdown","source":"**2:  Try to Find the best hyperparameters**"},{"metadata":{"trusted":true,"_uuid":"52c34ac4c88320ca0ea8c943efd82b176bffd59f"},"cell_type":"code","source":"\n## option 1\n\n# Number of trees in random forest\nn_estimators = [int(x) for x in np.linspace(start = 200, stop = 2000, num = 10)]\n# Number of features to consider at every split\nmax_features = ['auto', 'sqrt']\n# Maximum number of levels in tree\nmax_depth = [int(x) for x in np.linspace(10, 110, num = 11)]\nmax_depth.append(None)\n# Minimum number of samples required to split a node\nmin_samples_split = [2, 5, 10]\n# Minimum number of samples required at each leaf node\nmin_samples_leaf = [1, 2, 4]\n# Method of selecting samples for training each tree\nbootstrap = [True, False]\n# Create the random grid\nrandom_grid = {'n_estimators': n_estimators,\n               'max_features': max_features,\n               'max_depth': max_depth,\n               'min_samples_split': min_samples_split,\n               'min_samples_leaf': min_samples_leaf,\n               'bootstrap': bootstrap}\n\nprint(random_grid)\n\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"0b1879dca12169b0172cf3abfa3ee57770b54b8f"},"cell_type":"code","source":"## option 2\n'''\n\n# Number of trees in random forest\nn_estimators = [int(x) for x in np.linspace(start = 20, stop = 200, num = 5)]\n# Number of features to consider at every split\nmax_features = ['auto', 'sqrt']\n# Maximum number of levels in tree\nmax_depth = [int(x) for x in np.linspace(1, 45, num = 3)]\n#max_depth.append(None)\n\n# Minimum number of samples required to split a node\nmin_samples_split = [5, 10]\n# Minimum number of samples required at each leaf node\n#min_samples_leaf = [1, 2, 4]\n# Method of selecting samples for training each tree\n#bootstrap = [True, False]\n\n# Create the random grid\nrandom_grid = {'n_estimators': n_estimators,\n               'max_features': max_features,\n               'max_depth': max_depth,\n               'min_samples_split': min_samples_split}\n\nprint(random_grid)\n\n'''","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"37ecdabc23f1ac5d6db1892e4907050592a8a659"},"cell_type":"code","source":"# Use the random grid to search for best hyperparameters\n# First create the base model to tune\nrf = RandomForestRegressor()\n# Random search of parameters, using 3 fold cross validation, \n# search across 100 different combinations, and use all available cores\n\n##option 1\nrf_random = RandomizedSearchCV(estimator = rf, param_distributions = random_grid, n_iter = 100, cv = 3, verbose=2, random_state=42, n_jobs = -1)\n\n##option 2\n#rf_random = RandomizedSearchCV(estimator = rf, param_distributions = random_grid, n_iter = 10, cv = 10, verbose=2, random_state=42, n_jobs = -1, scoring='neg_mean_squared_error')\n\n\n# Fit the random search model\nrf_random.fit(train_X, train_y)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"c8defdf1755f5e5a8d9b096f07f7f80f052d3c11"},"cell_type":"markdown","source":"**2.1:  Display Selected Hyperparameters **"},{"metadata":{"trusted":true,"_uuid":"f22c1de99d1adf03f81668257a04ca9f58dc1af0"},"cell_type":"code","source":"print(rf_random.best_params_)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"717957ca840deb438f61799f7e08f45f5b54fafa"},"cell_type":"markdown","source":"**2.2:  Compare the Accuracy Between Basic RandomForest and the one with  RandomizedSearchCV and best Hyperparameters **"},{"metadata":{"trusted":true,"_uuid":"0267883b1c559e33ee486684d07e8c0f42aed008"},"cell_type":"code","source":"def evaluate(model, test_features, test_labels):\n    predictions = model.predict(test_features)\n    errors = abs(predictions - test_labels)\n    mape = 100 * np.mean(errors / test_labels)\n    accuracy = 100 - mape\n    print('Model Performance')\n    print('Average Error: {:0.4f} degrees.'.format(np.mean(errors)))\n    print('Accuracy = {:0.2f}%.'.format(accuracy))\n    print(' ')\n    return accuracy\n\nbase_model = RandomForestRegressor(n_estimators = 10, random_state = 42)\nbase_model.fit(train_X, train_y)\nbase_accuracy = evaluate(base_model,   val_X, val_y  )\n\n\n   \nbest_random = rf_random.best_estimator_\nrandom_accuracy = evaluate(best_random,  val_X, val_y  )\n\n\nprint('Improvement of {:0.2f}%.'.format( 100 * (random_accuracy - base_accuracy) / base_accuracy))\n","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"1e29da8a0097611646bc883b8f2d7267bf3e223a","trusted":true},"cell_type":"code","source":"# To improve accuracy, create a new Random Forest model which you will train on all training data\nrf_model_on_full_data =  rf_random.best_estimator_\n\n","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"413dc8034a641bc755222daa1a4c1aa04e15c9c3"},"cell_type":"markdown","source":"# Make Predictions\nRead the file of \"test\" data. And apply your model to make predictions"},{"metadata":{"_uuid":"2ced6d62b1523c8434ac45bd15c080a4cab94f2a","trusted":true},"cell_type":"code","source":"# path to file you will use for predictions\ntest_data_path = '../input/test.csv'\n\n# read test data file using pandas\ntest_data = pd.read_csv(test_data_path)\n\n# create test_X which comes from test_data but includes only the columns you used for prediction.\n# The list of columns is stored in a variable called features\n\n#Option 1: Basic Column Selection\n#featuresTest = ['LotArea', 'YearBuilt', '1stFlrSF', '2ndFlrSF', 'FullBath', 'BedroomAbvGr', 'TotRmsAbvGrd']\n#test_X = test_data[featuresTest]\n\n#Option 2: Only Numeric Columns\nfeaturesTest = ['MSSubClass', 'LotArea', 'OverallQual', 'OverallCond', 'YearBuilt', 'YearRemodAdd', 'BsmtFinSF1', 'BsmtFinSF2', 'BsmtUnfSF', 'TotalBsmtSF', '1stFlrSF', '2ndFlrSF', 'LowQualFinSF', 'GrLivArea', 'BsmtFullBath', 'BsmtHalfBath', 'FullBath', 'HalfBath', 'BedroomAbvGr', 'KitchenAbvGr', 'TotRmsAbvGrd', 'Fireplaces', 'GarageCars', 'GarageArea', 'WoodDeckSF', 'OpenPorchSF', 'EnclosedPorch', '3SsnPorch', 'ScreenPorch', 'PoolArea', 'MiscVal', 'MoSold', 'YrSold']\ntest_X = test_data[featuresTest]\ntest_X = test_X.fillna(test_X.mean())\n\ntest_preds = rf_model_on_full_data.predict(test_X)\n\n\n\nprint(test_preds)\n# The lines below shows how to save predictions in format used for competition scoring\n# Just uncomment them.\n\n\noutput = pd.DataFrame({'Id': test_data.Id,'SalePrice': test_preds})\noutput.to_csv('submission.csv', index=False)\nprint('CSV GENERATED !!!')","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}