{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nfrom sklearn.model_selection import train_test_split\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\n# Read the data\ndata_train = pd.read_csv('../input/house-prices-advanced-regression-techniques/train.csv', index_col='Id')\ndata_test_full = pd.read_csv('../input/house-prices-advanced-regression-techniques/test.csv', index_col='Id')","metadata":{"execution":{"iopub.status.busy":"2022-07-21T08:38:22.046071Z","iopub.execute_input":"2022-07-21T08:38:22.046619Z","iopub.status.idle":"2022-07-21T08:38:22.105634Z","shell.execute_reply.started":"2022-07-21T08:38:22.046560Z","shell.execute_reply":"2022-07-21T08:38:22.104620Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"var = 'GrLivArea'\ndata = pd.concat([data_train['SalePrice'], data_train[var]], axis=1)\ndata.plot.scatter(x=var, y='SalePrice', ylim=(0,800000));","metadata":{"execution":{"iopub.status.busy":"2022-07-21T08:38:22.858788Z","iopub.execute_input":"2022-07-21T08:38:22.859548Z","iopub.status.idle":"2022-07-21T08:38:23.092526Z","shell.execute_reply.started":"2022-07-21T08:38:22.859510Z","shell.execute_reply":"2022-07-21T08:38:23.090557Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"var = 'TotalBsmtSF'\ndata = pd.concat([data_train['SalePrice'], data_train[var]], axis=1)\ndata.plot.scatter(x=var, y='SalePrice', ylim=(0,800000));","metadata":{"execution":{"iopub.status.busy":"2022-07-21T08:38:23.808603Z","iopub.execute_input":"2022-07-21T08:38:23.809511Z","iopub.status.idle":"2022-07-21T08:38:24.018192Z","shell.execute_reply.started":"2022-07-21T08:38:23.809460Z","shell.execute_reply":"2022-07-21T08:38:24.016432Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"var = 'OverallQual'\ndata = pd.concat([data_train['SalePrice'], data_train[var]], axis=1)\nf, ax = plt.subplots(figsize=(8, 6))\nfig = sns.boxplot(x=var, y=\"SalePrice\", data=data)\nfig.axis(ymin=0, ymax=800000);","metadata":{"execution":{"iopub.status.busy":"2022-07-21T08:38:24.662510Z","iopub.execute_input":"2022-07-21T08:38:24.663745Z","iopub.status.idle":"2022-07-21T08:38:25.015637Z","shell.execute_reply.started":"2022-07-21T08:38:24.663699Z","shell.execute_reply":"2022-07-21T08:38:25.014621Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"var = 'YearBuilt'\ndata = pd.concat([data_train['SalePrice'], data_train[var]], axis=1)\nf, ax = plt.subplots(figsize=(16, 8))\nfig = sns.boxplot(x=var, y=\"SalePrice\", data=data)\nfig.axis(ymin=0, ymax=800000);\nplt.xticks(rotation=90);","metadata":{"execution":{"iopub.status.busy":"2022-07-21T08:38:25.558641Z","iopub.execute_input":"2022-07-21T08:38:25.560138Z","iopub.status.idle":"2022-07-21T08:38:28.383998Z","shell.execute_reply.started":"2022-07-21T08:38:25.560093Z","shell.execute_reply":"2022-07-21T08:38:28.382467Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# We choose features which seem to be related with 'SalePrice'\ny = data_train.SalePrice\nX = data_train[['GrLivArea', 'TotalBsmtSF', 'OverallQual', 'YearBuilt']]\n\n# Break off validation set from training data\nX_train_full, X_valid_full, y_train, y_valid = train_test_split(X, y, train_size=0.8, test_size=0.2, random_state=0)","metadata":{"execution":{"iopub.status.busy":"2022-07-21T08:38:29.418809Z","iopub.execute_input":"2022-07-21T08:38:29.419268Z","iopub.status.idle":"2022-07-21T08:38:29.431518Z","shell.execute_reply.started":"2022-07-21T08:38:29.419237Z","shell.execute_reply":"2022-07-21T08:38:29.429891Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.tree import DecisionTreeRegressor\nfrom sklearn.ensemble import RandomForestRegressor\nfrom sklearn.ensemble import AdaBoostClassifier\nfrom xgboost import XGBRegressor\nfrom sklearn.metrics import mean_absolute_error\n\n# Decision tree model\nbase_model = DecisionTreeRegressor(random_state=1)\nbase_model.fit(X_train_full, y_train)\npredict_1 = base_model.predict(X_valid_full)\nprint(f'MAE of Decision tree model: {mean_absolute_error(y_valid, predict_1)}, Score: {base_model.score(X_valid_full, y_valid)}')\n\n# Random forest model\nrandom_forest_model = RandomForestRegressor(random_state=1)\nrandom_forest_model.fit(X_train_full,y_train)\npredict_2 = random_forest_model.predict(X_valid_full)\nprint(f'MAE of Random forest model: {mean_absolute_error(y_valid, predict_2)}, Score: {random_forest_model.score(X_valid_full, y_valid)}')\n\n# Xgboost model\nxgboost_model = XGBRegressor(n_estimators=1000, learning_rate=0.01, n_jobs=-1, random_state=1)\nxgboost_model.fit(X_train_full, y_train)\npredict_3 = xgboost_model.predict(X_valid_full)\nprint(f'MAE of Xgboost model: {mean_absolute_error(y_valid, predict_3)}, Score: {xgboost_model.score(X_valid_full, y_valid)}')\n\n# Adaboost model\nadaboost_model = AdaBoostClassifier(n_estimators=1000, learning_rate=0.1, random_state=1)\nadaboost_model.fit(X_train_full, y_train)\npredict_4 = adaboost_model.predict(X_valid_full)\nprint(f'MAE of Adaboost model: {mean_absolute_error(y_valid, predict_4)}, Score: {adaboost_model.score(X_valid_full, y_valid)}')","metadata":{"execution":{"iopub.status.busy":"2022-07-21T08:38:30.958965Z","iopub.execute_input":"2022-07-21T08:38:30.959461Z","iopub.status.idle":"2022-07-21T08:39:09.970965Z","shell.execute_reply.started":"2022-07-21T08:38:30.959425Z","shell.execute_reply":"2022-07-21T08:39:09.969375Z"},"trusted":true},"execution_count":null,"outputs":[]}]}