{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-26T21:09:47.334939Z","iopub.execute_input":"2022-07-26T21:09:47.335358Z","iopub.status.idle":"2022-07-26T21:09:47.363894Z","shell.execute_reply.started":"2022-07-26T21:09:47.335271Z","shell.execute_reply":"2022-07-26T21:09:47.362819Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# read files\ntrain=pd.read_csv('../input/house-prices-advanced-regression-techniques/train.csv')\ntest=pd.read_csv('../input/house-prices-advanced-regression-techniques/test.csv')","metadata":{"execution":{"iopub.status.busy":"2022-07-26T21:10:36.973478Z","iopub.execute_input":"2022-07-26T21:10:36.973872Z","iopub.status.idle":"2022-07-26T21:10:37.024015Z","shell.execute_reply.started":"2022-07-26T21:10:36.973836Z","shell.execute_reply":"2022-07-26T21:10:37.022792Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Deal w/ missing values\nthere are many missing values in the dataset, we need to deal w/ them before applying random forest/xgboost","metadata":{}},{"cell_type":"code","source":"# take a look at features w/ missing values in training set\ntrain.isna().sum().sort_values(ascending=False).head(30)","metadata":{"execution":{"iopub.status.busy":"2022-07-26T21:10:39.383354Z","iopub.execute_input":"2022-07-26T21:10:39.384004Z","iopub.status.idle":"2022-07-26T21:10:39.402446Z","shell.execute_reply.started":"2022-07-26T21:10:39.383972Z","shell.execute_reply":"2022-07-26T21:10:39.401394Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# take a look at features w/ missing values in test set\ntest.isna().sum().sort_values(ascending=False).head(40)","metadata":{"execution":{"iopub.status.busy":"2022-07-26T21:10:43.925521Z","iopub.execute_input":"2022-07-26T21:10:43.925892Z","iopub.status.idle":"2022-07-26T21:10:43.943574Z","shell.execute_reply.started":"2022-07-26T21:10:43.925861Z","shell.execute_reply":"2022-07-26T21:10:43.942694Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# majority of Alley, PoolQC, Fence, MiscFeature, and FireplaceQu are missing values\n# these columns are not likely to provide much predictability \n# take out these columns\ntrain = train.drop(['Alley', 'PoolQC', 'Fence', 'MiscFeature','FireplaceQu'], axis=1)\ntest = test.drop(['Alley', 'PoolQC', 'Fence', 'MiscFeature','FireplaceQu'], axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-07-26T21:10:47.307011Z","iopub.execute_input":"2022-07-26T21:10:47.307921Z","iopub.status.idle":"2022-07-26T21:10:47.316154Z","shell.execute_reply.started":"2022-07-26T21:10:47.307887Z","shell.execute_reply":"2022-07-26T21:10:47.315189Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# replace missing values with the most common value in each column\nfor f in train.drop(['SalePrice'], axis=1).columns:\n    if train[f].isna().sum() > 0:\n        train[f].fillna(train[f].mode()[0], inplace=True)   \n    if test[f].isna().sum() > 0:\n        test[f].fillna(test[f].mode()[0], inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-26T21:10:53.065700Z","iopub.execute_input":"2022-07-26T21:10:53.066102Z","iopub.status.idle":"2022-07-26T21:10:53.153558Z","shell.execute_reply.started":"2022-07-26T21:10:53.066055Z","shell.execute_reply":"2022-07-26T21:10:53.152412Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Convert categorical variable into dummy/indicator variables\ntrain_dummy = pd.get_dummies(train, drop_first=True)\ntest_dummy = pd.get_dummies(test, drop_first=True)\n\npd.set_option('display.max_columns', 5)\nprint(train_dummy)\nprint('\\n', test_dummy)","metadata":{"execution":{"iopub.status.busy":"2022-07-26T21:11:14.580259Z","iopub.execute_input":"2022-07-26T21:11:14.580704Z","iopub.status.idle":"2022-07-26T21:11:14.663320Z","shell.execute_reply.started":"2022-07-26T21:11:14.580657Z","shell.execute_reply":"2022-07-26T21:11:14.662214Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"After converting categorical variables into dummy variables, we can see that the numbers of columns in training and test sets are not the same.\nThis indicates that there are columns in training (test) set w/ dummy variables that do not exist in test (training) set","metadata":{}},{"cell_type":"code","source":"# compare dummy variables in training and test sets\nprint('dummy variables not in test:')\nprint(list(set(train_dummy.drop(['SalePrice'], axis=1).columns.tolist()) - set(test_dummy.columns.tolist())))\nprint('\\ndummy variables not in training:')\nprint(list(set(test_dummy.columns.tolist()) - set(train_dummy.columns.tolist())))","metadata":{"execution":{"iopub.status.busy":"2022-07-26T21:12:27.012139Z","iopub.execute_input":"2022-07-26T21:12:27.012623Z","iopub.status.idle":"2022-07-26T21:12:27.028288Z","shell.execute_reply.started":"2022-07-26T21:12:27.012580Z","shell.execute_reply":"2022-07-26T21:12:27.026667Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# add dummy columns in training but not in test to 0\nfor c in list(set(train_dummy.drop(['SalePrice'], axis=1).columns.tolist()) - set(test_dummy.columns.tolist())):\n    test_dummy[c] = 0","metadata":{"execution":{"iopub.status.busy":"2022-07-26T21:12:35.006512Z","iopub.execute_input":"2022-07-26T21:12:35.006904Z","iopub.status.idle":"2022-07-26T21:12:35.020374Z","shell.execute_reply.started":"2022-07-26T21:12:35.006876Z","shell.execute_reply":"2022-07-26T21:12:35.019500Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Method 1: Random forest\nrandom forest is strong with outliers, irrelevant variables, continuous and discrete variables\nthe disadvantage in this case is the data is not big enough and random forest can easily overfit","metadata":{}},{"cell_type":"code","source":"from sklearn.ensemble import RandomForestRegressor\nfrom sklearn.metrics import mean_squared_error\n\n# build model\nrf = RandomForestRegressor(n_estimators=100, criterion='mse', max_depth=12, oob_score=True, #\\\n                          min_samples_leaf=2, min_samples_split=2, bootstrap=True)\nrf.fit(train_dummy.drop(['SalePrice','Id'], axis=1), train_dummy['SalePrice'])\n\n# take a look at mse of the training set\n# should should looked at test set instead if we had the sale prices, so we could adjust the parameters\npred1 = rf.predict(train_dummy.drop(['SalePrice','Id'], axis=1))\nmse1 = mean_squared_error(pred1, train_dummy['SalePrice'])\n\nprint(\"mse of train is \", mse1)","metadata":{"execution":{"iopub.status.busy":"2022-07-26T21:13:08.114536Z","iopub.execute_input":"2022-07-26T21:13:08.114921Z","iopub.status.idle":"2022-07-26T21:13:11.059745Z","shell.execute_reply.started":"2022-07-26T21:13:08.114889Z","shell.execute_reply":"2022-07-26T21:13:11.058174Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# predict sale prices\npred_rf = rf.predict(test_dummy.drop(['Id'], axis=1))","metadata":{"execution":{"iopub.status.busy":"2022-07-26T21:13:15.552034Z","iopub.execute_input":"2022-07-26T21:13:15.553037Z","iopub.status.idle":"2022-07-26T21:13:15.602581Z","shell.execute_reply.started":"2022-07-26T21:13:15.552985Z","shell.execute_reply":"2022-07-26T21:13:15.601453Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Method 2: XGBoost\nXGBoost is more powerful than random forest\nbut again, disadvantage here is the small dataset","metadata":{}},{"cell_type":"code","source":"from xgboost import XGBRegressor\nfrom sklearn.metrics import mean_squared_error\n\n# build model\nxgbr = XGBRegressor(booster='gbtree', learning_rate=0.3,\n        max_depth=5, n_estimators=100, eval_metric='rmse',\n       objective='reg:squarederror', reg_alpha=0,\n       reg_lambda=0, subsample=1, verbosity=1) ","metadata":{"execution":{"iopub.status.busy":"2022-07-26T21:13:18.203884Z","iopub.execute_input":"2022-07-26T21:13:18.204372Z","iopub.status.idle":"2022-07-26T21:13:18.351056Z","shell.execute_reply.started":"2022-07-26T21:13:18.204331Z","shell.execute_reply":"2022-07-26T21:13:18.350185Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# fit model on training data\nxgbr.fit(train_dummy.drop(['SalePrice','Id'], axis=1), train_dummy['SalePrice'])\nprint(xgbr)","metadata":{"execution":{"iopub.status.busy":"2022-07-26T21:13:23.272413Z","iopub.execute_input":"2022-07-26T21:13:23.273067Z","iopub.status.idle":"2022-07-26T21:13:24.207134Z","shell.execute_reply.started":"2022-07-26T21:13:23.273033Z","shell.execute_reply":"2022-07-26T21:13:24.206145Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# take a look at mse of the training set\n# should should looked at test set instead if we had the sale prices, so we could adjust the parameters\nprint(mean_squared_error(xgbr.predict(train_dummy.drop(['SalePrice','Id'], axis=1)), train_dummy['SalePrice']))\n\npred_xg = xgbr.predict(test_dummy.drop(['Id'], axis=1))","metadata":{"execution":{"iopub.status.busy":"2022-07-26T21:13:38.640409Z","iopub.execute_input":"2022-07-26T21:13:38.640815Z","iopub.status.idle":"2022-07-26T21:13:38.685318Z","shell.execute_reply.started":"2022-07-26T21:13:38.640780Z","shell.execute_reply":"2022-07-26T21:13:38.684226Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# save to file\nout = pd.DataFrame({'Id': test_dummy['Id'],\n                   'SalePrice': pred_xg})\nout.to_csv('test_SalePrice.cvs', header=True, index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-26T20:04:03.658901Z","iopub.execute_input":"2022-07-26T20:04:03.659470Z","iopub.status.idle":"2022-07-26T20:04:03.679312Z","shell.execute_reply.started":"2022-07-26T20:04:03.659416Z","shell.execute_reply":"2022-07-26T20:04:03.677312Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}