{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# Import helpful libraries\nimport pandas as pd\nimport numpy as np\nfrom sklearn.ensemble import RandomForestRegressor\nfrom sklearn.metrics import mean_absolute_error, mean_squared_error\nfrom sklearn.model_selection import train_test_split\n\n# Load the data, and separate the target\niowa_file_path = '../input/home-data-for-ml-course/train.csv'\nhome_data = pd.read_csv(iowa_file_path)\ny = home_data.SalePrice","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-02T00:19:55.211558Z","iopub.execute_input":"2022-08-02T00:19:55.212128Z","iopub.status.idle":"2022-08-02T00:19:56.375732Z","shell.execute_reply.started":"2022-08-02T00:19:55.212049Z","shell.execute_reply":"2022-08-02T00:19:56.374865Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Get a count of missing values per column.\nmissing_values_count = home_data.isnull().sum()\nmissing_values_count[0:25]\n\n# There are a ton of rows missing Alley, probably because they don't have alleys. Drop this feature\n# There are also some missing LotFrontage. ","metadata":{"execution":{"iopub.status.busy":"2022-08-02T00:20:01.218323Z","iopub.execute_input":"2022-08-02T00:20:01.218743Z","iopub.status.idle":"2022-08-02T00:20:01.235839Z","shell.execute_reply.started":"2022-08-02T00:20:01.218711Z","shell.execute_reply":"2022-08-02T00:20:01.235057Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Filling lot frontage with the column mean so we can use that feature.\nlotfrontage_mean = home_data['LotFrontage'].mean()\nhome_data['LotFrontage'].fillna(value=lotfrontage_mean, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-02T00:20:07.161631Z","iopub.execute_input":"2022-08-02T00:20:07.161952Z","iopub.status.idle":"2022-08-02T00:20:07.167180Z","shell.execute_reply.started":"2022-08-02T00:20:07.161928Z","shell.execute_reply":"2022-08-02T00:20:07.165789Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Checking to see if the fill worked. It did.\n# We aren't going to worry about Alley because it's missing so many\nmissing_values_count = home_data.isnull().sum()\nmissing_values_count[50:]","metadata":{"execution":{"iopub.status.busy":"2022-08-02T00:22:40.747022Z","iopub.execute_input":"2022-08-02T00:22:40.747355Z","iopub.status.idle":"2022-08-02T00:22:40.759029Z","shell.execute_reply.started":"2022-08-02T00:22:40.747328Z","shell.execute_reply":"2022-08-02T00:22:40.758331Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# this is just a list of all features (columns) in the dataset\n'MSSubClass', 'LotArea', 'LotFrontage', 'OverallQual', 'OverallCond', 'YearBuilt', 'YearRemodAdd', '1stFlrSF', '2ndFlrSF',\n'LowQualFinSF', 'GrLivArea', 'FullBath', 'HalfBath', 'BedroomAbvGr', 'KitchenAbvGr', 'TotRmsAbvGrd',\n'Fireplaces', 'WoodDeckSF', 'OpenPorchSF', 'EnclosedPorch', '3SsnPorch', 'ScreenPorch', 'PoolArea',\n'MiscVal', 'MoSold', 'YrSold'","metadata":{"execution":{"iopub.status.busy":"2022-08-01T23:57:30.822233Z","iopub.execute_input":"2022-08-01T23:57:30.822617Z","iopub.status.idle":"2022-08-01T23:57:30.831206Z","shell.execute_reply.started":"2022-08-01T23:57:30.822587Z","shell.execute_reply":"2022-08-01T23:57:30.829836Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Original Feature list\n\nfeatures = ['LotArea', 'YearBuilt', '1stFlrSF', '2ndFlrSF', 'FullBath', 'BedroomAbvGr', \n            'TotRmsAbvGrd']\n\nGave an RMSE of 31,470.\n\nI'm going to add all features, and start slowly removing them to see what improves the model.","metadata":{}},{"cell_type":"markdown","source":"Model features:\n\nfeatures = ['MSSubClass', 'LotArea', 'LotFrontage', 'OverallQual', 'OverallCond', 'YearBuilt', 'YearRemodAdd', '1stFlrSF', '2ndFlrSF',\n            'LowQualFinSF', 'GrLivArea', 'FullBath', 'HalfBath', 'BedroomAbvGr', 'KitchenAbvGr', 'TotRmsAbvGrd',\n            'Fireplaces', 'WoodDeckSF', 'OpenPorchSF', 'EnclosedPorch', '3SsnPorch', 'ScreenPorch', 'PoolArea',\n            'MiscVal', 'MoSold', 'YrSold']\n\n\nRMSE: 27,364\n\nAlready an improvement.\n\nI'm going to remove some features after looking at the description of the variables on the data tab of the page. The number after is the RMSE after removing\n\n* MSSubClass: The building class (27,375)\n\n* OverallQual: Overall material and finish quality (29,407) We seem to be going the wrong direction...\n\n* 'MiscVal', 'MoSold', 'YrSold'(28,995)\n\n* '1stFlrSF', '2ndFlrSF' (33,228)\n\n* 'LowQualFinSF' (33,495)\n\n*  'YearRemodAdd' (33,415)\n\n* 'GrLivArea', (38,842)\n\n* 'WoodDeckSF', 'OpenPorchSF', 'EnclosedPorch', '3SsnPorch', 'ScreenPorch', 'PoolArea' (41,031)\n\n* Adding 'WoodDeckSF', 'OpenPorchSF', 'EnclosedPorch', '3SsnPorch', 'ScreenPorch' in (38,994)\n\n* 'Fireplaces' (40,972)\n\n* 'LotFrontage' (39,041)","metadata":{}},{"cell_type":"code","source":"# Create X (After completing the exercise, you can return to modify this line!)\nfeatures = ['MSSubClass', 'LotArea', 'OverallQual', 'OverallCond', 'YearBuilt', 'YearRemodAdd', '1stFlrSF', '2ndFlrSF',\n'LowQualFinSF', 'GrLivArea', 'FullBath', 'HalfBath', 'BedroomAbvGr', 'KitchenAbvGr', 'TotRmsAbvGrd',\n'Fireplaces', 'WoodDeckSF', 'OpenPorchSF', 'EnclosedPorch', '3SsnPorch', 'ScreenPorch', 'PoolArea',\n'MiscVal', 'MoSold', 'YrSold']\n\n# Select columns corresponding to features, and preview the data\nX = home_data[features]\n\n# Split into validation and training data\ntrain_X, val_X, train_y, val_y = train_test_split(X, y, random_state=1)\n\n# Define a random forest model\nrf_model = RandomForestRegressor(random_state=1, n_estimators=700)\nrf_model.fit(train_X, train_y)\nrf_val_predictions = rf_model.predict(val_X)\nrf_val_mae = mean_absolute_error(rf_val_predictions, val_y)\nrf_val_rmse = np.sqrt(mean_squared_error(rf_val_predictions, val_y)) # Added to get the RMSE which is what the competition is based on.\n\nprint(\"Validation MAE for Random Forest Model: {:,.0f}\".format(rf_val_mae))\nprint(f\"Validation RMSE for Random Forest Model: {rf_val_rmse}\")","metadata":{"execution":{"iopub.status.busy":"2022-08-02T00:23:15.444021Z","iopub.execute_input":"2022-08-02T00:23:15.444377Z","iopub.status.idle":"2022-08-02T00:23:19.890363Z","shell.execute_reply.started":"2022-08-02T00:23:15.444348Z","shell.execute_reply":"2022-08-02T00:23:19.889198Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Other model results\n\nValidation MAE for Random Forest Model: 21,857\n\nValidation RMSE for Random Forest Model: 31619.525018415294","metadata":{}},{"cell_type":"code","source":"# To improve accuracy, create a new Random Forest model which you will train on all training data\nrf_model_on_full_data = RandomForestRegressor(random_state=1, n_estimators=700)\n\n# fit rf_model_on_full_data on all data from the training data\nrf_model_on_full_data.fit(X, y)","metadata":{"execution":{"iopub.status.busy":"2022-08-02T00:23:27.192909Z","iopub.execute_input":"2022-08-02T00:23:27.193612Z","iopub.status.idle":"2022-08-02T00:23:32.968708Z","shell.execute_reply.started":"2022-08-02T00:23:27.193583Z","shell.execute_reply":"2022-08-02T00:23:32.968045Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# path to file you will use for predictions\ntest_data_path = '../input/home-data-for-ml-course/test.csv'\n\n# read test data file using pandas\ntest_data = pd.read_csv(test_data_path)\n\n# create test_X which comes from test_data but includes only the columns you used for prediction.\n# The list of columns is stored in a variable called features\ntest_X = test_data[features]\n\n# make predictions which we will submit. \ntest_preds = rf_model_on_full_data.predict(test_X)","metadata":{"execution":{"iopub.status.busy":"2022-08-02T00:23:36.523901Z","iopub.execute_input":"2022-08-02T00:23:36.524239Z","iopub.status.idle":"2022-08-02T00:23:36.737678Z","shell.execute_reply.started":"2022-08-02T00:23:36.524213Z","shell.execute_reply":"2022-08-02T00:23:36.736406Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Run the code to save predictions in the format used for competition scoring\n\noutput = pd.DataFrame({'Id': test_data.Id,\n                       'SalePrice': test_preds})\noutput.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-08-02T00:23:39.384656Z","iopub.execute_input":"2022-08-02T00:23:39.385531Z","iopub.status.idle":"2022-08-02T00:23:39.396244Z","shell.execute_reply.started":"2022-08-02T00:23:39.385497Z","shell.execute_reply":"2022-08-02T00:23:39.395420Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}