{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-07T15:09:44.816138Z","iopub.execute_input":"2022-08-07T15:09:44.817229Z","iopub.status.idle":"2022-08-07T15:09:44.826944Z","shell.execute_reply.started":"2022-08-07T15:09:44.817181Z","shell.execute_reply":"2022-08-07T15:09:44.825712Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport seaborn as sns\nimport numpy as np\nimport matplotlib.pyplot as plt\nfrom scipy.stats import norm\n\nimport random\n\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.linear_model import LinearRegression\nfrom sklearn.tree import DecisionTreeRegressor\nfrom sklearn.ensemble import RandomForestRegressor\nfrom sklearn.model_selection import train_test_split, cross_val_score\nfrom sklearn.metrics import mean_absolute_error, r2_score, mean_squared_error\nfrom sklearn.utils import shuffle\nfrom xgboost import XGBRegressor\nfrom sklearn.metrics import accuracy_score\n\nimport warnings\nwarnings.filterwarnings('ignore')","metadata":{"execution":{"iopub.status.busy":"2022-08-07T15:09:44.897448Z","iopub.execute_input":"2022-08-07T15:09:44.898203Z","iopub.status.idle":"2022-08-07T15:09:44.906644Z","shell.execute_reply.started":"2022-08-07T15:09:44.898166Z","shell.execute_reply":"2022-08-07T15:09:44.905724Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test = pd.read_csv('../input/house-prices-advanced-regression-techniques/test.csv')\ndf = pd.read_csv('../input/house-prices-advanced-regression-techniques/train.csv')\ndf","metadata":{"execution":{"iopub.status.busy":"2022-08-07T15:09:44.981215Z","iopub.execute_input":"2022-08-07T15:09:44.982292Z","iopub.status.idle":"2022-08-07T15:09:45.051930Z","shell.execute_reply.started":"2022-08-07T15:09:44.982239Z","shell.execute_reply":"2022-08-07T15:09:45.050734Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.columns, df.SalePrice, df.info()","metadata":{"execution":{"iopub.status.busy":"2022-08-07T15:09:45.092704Z","iopub.execute_input":"2022-08-07T15:09:45.093117Z","iopub.status.idle":"2022-08-07T15:09:45.117865Z","shell.execute_reply.started":"2022-08-07T15:09:45.093080Z","shell.execute_reply":"2022-08-07T15:09:45.116818Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"obj_cols = [cols for cols in df.columns if\n           df[cols].dtype == \"object\"]\n\nfeat_cols = [cols for cols in df.columns if\n           df[cols].dtype == \"int64\"]\nfeat_cols.pop(34)\nfeat_cols","metadata":{"execution":{"iopub.status.busy":"2022-08-07T15:09:45.166812Z","iopub.execute_input":"2022-08-07T15:09:45.167225Z","iopub.status.idle":"2022-08-07T15:09:45.177141Z","shell.execute_reply.started":"2022-08-07T15:09:45.167188Z","shell.execute_reply":"2022-08-07T15:09:45.175971Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = feat_cols\ny = df[\"SalePrice\"]","metadata":{"execution":{"iopub.status.busy":"2022-08-07T15:09:45.238093Z","iopub.execute_input":"2022-08-07T15:09:45.238647Z","iopub.status.idle":"2022-08-07T15:09:45.243215Z","shell.execute_reply.started":"2022-08-07T15:09:45.238607Z","shell.execute_reply":"2022-08-07T15:09:45.242116Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Data Visualization**","metadata":{}},{"cell_type":"code","source":"sns.set_theme(style=\"ticks\")\n\nvar = 'GrLivArea'\ndata = pd.concat([df['SalePrice'], df[var]], axis=1)\ndata.plot.scatter(x=var, y='SalePrice', marker = (5, 0), ylim=(0,800000));","metadata":{"execution":{"iopub.status.busy":"2022-08-07T15:09:45.309272Z","iopub.execute_input":"2022-08-07T15:09:45.310547Z","iopub.status.idle":"2022-08-07T15:09:45.499998Z","shell.execute_reply.started":"2022-08-07T15:09:45.310495Z","shell.execute_reply":"2022-08-07T15:09:45.498952Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"var = 'TotalBsmtSF'\ndata = pd.concat([df['SalePrice'], df[var]], axis=1)\ndata.plot.scatter(x=var, y='SalePrice', ylim=(0,800000));","metadata":{"execution":{"iopub.status.busy":"2022-08-07T15:09:45.501930Z","iopub.execute_input":"2022-08-07T15:09:45.502270Z","iopub.status.idle":"2022-08-07T15:09:45.706755Z","shell.execute_reply.started":"2022-08-07T15:09:45.502239Z","shell.execute_reply":"2022-08-07T15:09:45.705605Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"var = 'OverallQual'\ndata = pd.concat([df['SalePrice'], df[var]], axis=1)\nf, ax = plt.subplots(figsize=(8, 6))\nfig = sns.boxplot(x=var, y=\"SalePrice\", data=data)\nfig.axis(ymin=0, ymax=800000);","metadata":{"execution":{"iopub.status.busy":"2022-08-07T15:09:45.708193Z","iopub.execute_input":"2022-08-07T15:09:45.708553Z","iopub.status.idle":"2022-08-07T15:09:46.014052Z","shell.execute_reply.started":"2022-08-07T15:09:45.708520Z","shell.execute_reply":"2022-08-07T15:09:46.012879Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"var = 'YearBuilt'\ndata = pd.concat([df['SalePrice'], df[var]], axis=1)\nf, ax = plt.subplots(figsize=(16, 8))\nfig = sns.boxplot(x=var, y=\"SalePrice\", data=data)\nfig.axis(ymin=0, ymax=800000);\nplt.xticks(rotation=90);","metadata":{"execution":{"iopub.status.busy":"2022-08-07T15:09:46.016596Z","iopub.execute_input":"2022-08-07T15:09:46.016941Z","iopub.status.idle":"2022-08-07T15:09:48.587703Z","shell.execute_reply.started":"2022-08-07T15:09:46.016909Z","shell.execute_reply":"2022-08-07T15:09:48.586658Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#correlation matrix\ncorrmat = df.corr()\nf, ax = plt.subplots(figsize=(12, 9))\nsns.heatmap(corrmat, vmax=.8, square=True);","metadata":{"execution":{"iopub.status.busy":"2022-08-07T15:09:48.589056Z","iopub.execute_input":"2022-08-07T15:09:48.589653Z","iopub.status.idle":"2022-08-07T15:09:49.516642Z","shell.execute_reply.started":"2022-08-07T15:09:48.589618Z","shell.execute_reply":"2022-08-07T15:09:49.515595Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"k = 10 #number of variables for heatmap\ncols = corrmat.nlargest(k, 'SalePrice')['SalePrice'].index\ncm = np.corrcoef(df[cols].values.T)\nsns.set(font_scale=1.25)\nhm = sns.heatmap(cm, cbar=True, annot=True, square=True, fmt='.2f', annot_kws={'size': 10}, yticklabels=cols.values, xticklabels=cols.values)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-07T15:09:49.518225Z","iopub.execute_input":"2022-08-07T15:09:49.518676Z","iopub.status.idle":"2022-08-07T15:09:50.223247Z","shell.execute_reply.started":"2022-08-07T15:09:49.518643Z","shell.execute_reply":"2022-08-07T15:09:50.222079Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#scatterplot\nsns.set()\ncols = ['SalePrice', 'OverallQual', 'OverallCond', 'GrLivArea', 'GarageCars', 'TotalBsmtSF', 'FullBath', 'YearBuilt']\nsns.pairplot(df[cols], height = 2.5)\nplt.show();","metadata":{"execution":{"iopub.status.busy":"2022-08-07T15:09:50.224773Z","iopub.execute_input":"2022-08-07T15:09:50.227751Z","iopub.status.idle":"2022-08-07T15:10:02.039292Z","shell.execute_reply.started":"2022-08-07T15:09:50.227699Z","shell.execute_reply":"2022-08-07T15:10:02.038422Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Missing Data**","metadata":{}},{"cell_type":"code","source":"#missing data\ntotal = df.isnull().sum().sort_values(ascending=False)\npercent = (df.isnull().sum()/df.isnull().count()).sort_values(ascending=False)\nmissing_data = pd.concat([total, percent], axis=1, keys=['Total', 'Percent'])\nmissing_data.head(20)","metadata":{"execution":{"iopub.status.busy":"2022-08-07T15:10:02.040531Z","iopub.execute_input":"2022-08-07T15:10:02.041552Z","iopub.status.idle":"2022-08-07T15:10:02.073025Z","shell.execute_reply.started":"2022-08-07T15:10:02.041515Z","shell.execute_reply":"2022-08-07T15:10:02.071909Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## **TO BE DONE**","metadata":{}},{"cell_type":"code","source":"#dealing with missing data\ndf2 = df.drop((missing_data[missing_data['Total'] > 1]).index,1)\n# df2 = df.drop(df.loc[df['Electrical'].isnull()].index)\ndf2.isnull().sum().max() #just checking that there's no missing data missing...","metadata":{"execution":{"iopub.status.busy":"2022-08-07T15:10:02.076399Z","iopub.execute_input":"2022-08-07T15:10:02.076827Z","iopub.status.idle":"2022-08-07T15:10:02.090581Z","shell.execute_reply.started":"2022-08-07T15:10:02.076796Z","shell.execute_reply":"2022-08-07T15:10:02.089639Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Analysing the OutLiars**\n\n* ## **UniVariant Analysis**\nThe primary concern here is to establish a threshold that defines an observation as an outlier. To do so, we'll standardize the data. In this context, data standardization means converting data values to have mean of 0 and a standard deviation of 1.","metadata":{}},{"cell_type":"code","source":"std_scaler = StandardScaler().fit_transform (df2['SalePrice'][:, np.newaxis]);\nl_ran = std_scaler[std_scaler[:,0].argsort()][:10]\nh_ran = std_scaler[std_scaler[:,0].argsort()][-10:]\n\nprint(\"Low Range[Outer]:\\n\\n\", l_ran)\nprint(\"Low Range[Outer]:\\n\\n\", h_ran)","metadata":{"execution":{"iopub.status.busy":"2022-08-07T15:10:02.091996Z","iopub.execute_input":"2022-08-07T15:10:02.092848Z","iopub.status.idle":"2022-08-07T15:10:02.102054Z","shell.execute_reply.started":"2022-08-07T15:10:02.092811Z","shell.execute_reply":"2022-08-07T15:10:02.100791Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* ## **Bivariate analysis**\n        We already know the following scatter plots by heart. However, when we look to things from a new perspective, there's\n        always something to discover. As Alan Kay said, 'a change in perspective is worth 80 IQ points'.","metadata":{}},{"cell_type":"code","source":"# bivariate analysis saleprice/grlivarea\nvar = 'GrLivArea'\ndata = pd.concat([df2['SalePrice'], df2[var]], axis=1)\ndata.plot.scatter(x=var, y='SalePrice', ylim=(0,800000));","metadata":{"execution":{"iopub.status.busy":"2022-08-07T15:10:02.103579Z","iopub.execute_input":"2022-08-07T15:10:02.104591Z","iopub.status.idle":"2022-08-07T15:10:02.354769Z","shell.execute_reply.started":"2022-08-07T15:10:02.104546Z","shell.execute_reply":"2022-08-07T15:10:02.353654Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## deleting points\ndf2.sort_values(by = 'GrLivArea', ascending = False)[:2]\ndf3 = df2.drop(df2[df2['Id'] == 1299].index)\ndf3 = df2.drop(df2[df2['Id'] == 524].index)\n\n#bivariate analysis saleprice/grlivarea\nvar = 'TotalBsmtSF'\ndata = pd.concat([df2['SalePrice'], df2[var]], axis=1)\ndata.plot.scatter(x=var, y='SalePrice', ylim=(0,800000));","metadata":{"execution":{"iopub.status.busy":"2022-08-07T15:10:02.356129Z","iopub.execute_input":"2022-08-07T15:10:02.356483Z","iopub.status.idle":"2022-08-07T15:10:02.616337Z","shell.execute_reply.started":"2022-08-07T15:10:02.356450Z","shell.execute_reply":"2022-08-07T15:10:02.615267Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#histogram and normal probability plot\nsns.distplot(df3['SalePrice'], fit=norm)\nfig = plt.figure()\n# res = stats.probplot(df3['SalePrice'], plot=plt)","metadata":{"execution":{"iopub.status.busy":"2022-08-07T15:10:02.618156Z","iopub.execute_input":"2022-08-07T15:10:02.618999Z","iopub.status.idle":"2022-08-07T15:10:02.952329Z","shell.execute_reply.started":"2022-08-07T15:10:02.618952Z","shell.execute_reply":"2022-08-07T15:10:02.951252Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df3['SalePrice'] = np.log(df3['SalePrice'])","metadata":{"execution":{"iopub.status.busy":"2022-08-07T15:10:02.953794Z","iopub.execute_input":"2022-08-07T15:10:02.954578Z","iopub.status.idle":"2022-08-07T15:10:02.961715Z","shell.execute_reply.started":"2022-08-07T15:10:02.954543Z","shell.execute_reply":"2022-08-07T15:10:02.960259Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.distplot(df3['SalePrice'], fit=norm)\nfig = plt.figure()\n# res = stats.probplot(df['SalePrice'], plot=plt)","metadata":{"execution":{"iopub.status.busy":"2022-08-07T15:10:02.962741Z","iopub.execute_input":"2022-08-07T15:10:02.963046Z","iopub.status.idle":"2022-08-07T15:10:03.287650Z","shell.execute_reply.started":"2022-08-07T15:10:02.963017Z","shell.execute_reply":"2022-08-07T15:10:03.286594Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#scatter plot\nplt.scatter(df3['GrLivArea'], df3['SalePrice']);","metadata":{"execution":{"iopub.status.busy":"2022-08-07T15:10:03.289083Z","iopub.execute_input":"2022-08-07T15:10:03.289461Z","iopub.status.idle":"2022-08-07T15:10:03.518232Z","shell.execute_reply.started":"2022-08-07T15:10:03.289428Z","shell.execute_reply":"2022-08-07T15:10:03.516954Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#scatter plot\nplt.scatter(df3[df3['TotalBsmtSF']>0]['TotalBsmtSF'], df3[df3['TotalBsmtSF']>0]['SalePrice']);","metadata":{"execution":{"iopub.status.busy":"2022-08-07T15:10:03.519730Z","iopub.execute_input":"2022-08-07T15:10:03.520174Z","iopub.status.idle":"2022-08-07T15:10:03.763298Z","shell.execute_reply.started":"2022-08-07T15:10:03.520142Z","shell.execute_reply":"2022-08-07T15:10:03.762404Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"__________________________________________________________________________________________","metadata":{}},{"cell_type":"code","source":"features = ['GrLivArea', 'LotArea', 'BedroomAbvGr', '1stFlrSF', '2ndFlrSF', 'TotRmsAbvGrd']\nX = df3[features]\ny = df3['SalePrice']","metadata":{"execution":{"iopub.status.busy":"2022-08-07T15:10:03.764437Z","iopub.execute_input":"2022-08-07T15:10:03.765450Z","iopub.status.idle":"2022-08-07T15:10:03.772204Z","shell.execute_reply.started":"2022-08-07T15:10:03.765399Z","shell.execute_reply":"2022-08-07T15:10:03.771421Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# TrainTestSPlit\ntrain_X, val_X, train_y, val_y = train_test_split(X.values, y.values, test_size = 0.3, random_state = 2)","metadata":{"execution":{"iopub.status.busy":"2022-08-07T15:10:03.773540Z","iopub.execute_input":"2022-08-07T15:10:03.774410Z","iopub.status.idle":"2022-08-07T15:10:03.783530Z","shell.execute_reply.started":"2022-08-07T15:10:03.774347Z","shell.execute_reply":"2022-08-07T15:10:03.782445Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"my_imputer = SimpleImputer()\ntrain_X = my_imputer.fit_transform(train_X)\nval_X = my_imputer.transform(val_X)","metadata":{"execution":{"iopub.status.busy":"2022-08-07T15:10:03.784614Z","iopub.execute_input":"2022-08-07T15:10:03.785180Z","iopub.status.idle":"2022-08-07T15:10:03.794661Z","shell.execute_reply.started":"2022-08-07T15:10:03.785146Z","shell.execute_reply":"2022-08-07T15:10:03.793609Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Data Handling Categorical Values**","metadata":{}},{"cell_type":"code","source":"# from sklearn.preprocessing import OneHotEncoder\n\n# OH_encoder = OneHotEncoder(handle_unknown = 'ignore', sparse = False)\n# OH_cols_train = pd.DataFrame(OH_encoder.fit_transform(train_X[obj_cols]))\n# OH_train_cols = pd.DataFrame(OH_encoder.transform(val_X[obj_cols]))\n\n# OH_cols_train.index = train_X.index\n# OH_cols_valid.index = val_X.index\n\n# num_X_train = train_X.drop(object_cols, axis=1)\n# num_X_valid = val_X.drop(object_cols, axis=1)\n\n# OH_X_train = pd.concat([num_X_train, OH_cols_train], axis=1)\n# OH_X_valid = pd.concat([num_X_valid, OH_cols_valid], axis=1)\n\n# print(\"MAE from Approach 3 (One-Hot Encoding):\") \n# print(score_dataset(OH_X_train, OH_X_valid, train_y, val_y))","metadata":{"execution":{"iopub.status.busy":"2022-08-07T15:10:03.798835Z","iopub.execute_input":"2022-08-07T15:10:03.799527Z","iopub.status.idle":"2022-08-07T15:10:03.804235Z","shell.execute_reply.started":"2022-08-07T15:10:03.799491Z","shell.execute_reply":"2022-08-07T15:10:03.803088Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"n_folds = 5\nfrom sklearn.metrics import make_scorer\nfrom sklearn.model_selection import KFold\nscorer = make_scorer(mean_squared_error,greater_is_better = False)\ndef rmse_CV_train(model):\n    kf = KFold(n_folds,shuffle=True,random_state=42).get_n_splits(df3.values)\n    rmse = np.sqrt(-cross_val_score(model, train_X, train_y,scoring =\"neg_mean_squared_error\",cv=kf))\n    return (rmse)\ndef rmse_CV_test(model):\n    kf = KFold(n_folds,shuffle=True,random_state=42).get_n_splits(df3.values)\n    rmse = np.sqrt(-cross_val_score(model,val_X,val_y,scoring =\"neg_mean_squared_error\",cv=kf))\n    return (rmse)","metadata":{"execution":{"iopub.status.busy":"2022-08-07T15:10:03.805800Z","iopub.execute_input":"2022-08-07T15:10:03.806560Z","iopub.status.idle":"2022-08-07T15:10:03.819019Z","shell.execute_reply.started":"2022-08-07T15:10:03.806516Z","shell.execute_reply":"2022-08-07T15:10:03.818100Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## LinearRegression","metadata":{}},{"cell_type":"code","source":"## Linear Regression\nmodel_lr = LinearRegression()\nmodel_lr.fit(train_X, train_y)\npred_lr= model_lr.predict(val_X)\nval_mae_lr = mean_absolute_error(pred_lr, val_y)\n\nprint(\"Prediction MEan Error\", (val_mae_lr)*100)","metadata":{"execution":{"iopub.status.busy":"2022-08-07T15:10:03.820601Z","iopub.execute_input":"2022-08-07T15:10:03.821254Z","iopub.status.idle":"2022-08-07T15:10:03.834067Z","shell.execute_reply.started":"2022-08-07T15:10:03.821219Z","shell.execute_reply":"2022-08-07T15:10:03.833189Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_f = model_lr.predict(val_X)\ntrain_f = model_lr.predict(train_X)\n\nprint('rmse on trainset: ', rmse_CV_train(model_lr).mean())\nprint('rmse on testset: ', rmse_CV_test(model_lr).mean())","metadata":{"execution":{"iopub.status.busy":"2022-08-07T15:10:03.838766Z","iopub.execute_input":"2022-08-07T15:10:03.839724Z","iopub.status.idle":"2022-08-07T15:10:03.866553Z","shell.execute_reply.started":"2022-08-07T15:10:03.839689Z","shell.execute_reply":"2022-08-07T15:10:03.865337Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## RandomForest","metadata":{}},{"cell_type":"code","source":"## Random Forest\nmodel_rf = RandomForestRegressor(n_estimators=200)\nmodel_rf.fit(train_X, train_y)\npred_rf = model_rf.predict(val_X)\nrf_val_mae = mean_absolute_error(pred_rf, val_y)\n\n# print(\"Validation MAE for Random Forest Model: {:,.0f}\".format(val_mae_dt))\nprint(\"Predictions Random Forest\", rf_val_mae * 100)","metadata":{"execution":{"iopub.status.busy":"2022-08-07T15:10:03.868316Z","iopub.execute_input":"2022-08-07T15:10:03.869109Z","iopub.status.idle":"2022-08-07T15:10:04.634623Z","shell.execute_reply.started":"2022-08-07T15:10:03.869063Z","shell.execute_reply":"2022-08-07T15:10:04.633572Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## XGBoost","metadata":{}},{"cell_type":"code","source":"model_xgb = XGBRegressor(n_estimators=5000)\nmodel_xgb.fit(train_X, train_y, early_stopping_rounds = 5, eval_set =[(val_X, val_y)], verbose=False)\npred_xgb = model_xgb.predict(val_X)\n\n# print(\"Predictions XGBoost:\", accuracy_score(model_xgb, val_y)) ## TO BE CHECKED FOR FINE-TUNING","metadata":{"execution":{"iopub.status.busy":"2022-08-07T15:10:04.635808Z","iopub.execute_input":"2022-08-07T15:10:04.636113Z","iopub.status.idle":"2022-08-07T15:10:04.969901Z","shell.execute_reply.started":"2022-08-07T15:10:04.636084Z","shell.execute_reply":"2022-08-07T15:10:04.968947Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_f1 = model_xgb.predict(val_X)\ntrain_f1 = model_xgb.predict(train_X)\n\nprint('rmse on trainset: ', rmse_CV_train(model_xgb).mean())\nprint('rmse on testset: ', rmse_CV_test(model_xgb).mean())","metadata":{"execution":{"iopub.status.busy":"2022-08-07T15:10:04.971431Z","iopub.execute_input":"2022-08-07T15:10:04.972719Z","iopub.status.idle":"2022-08-07T15:13:06.624535Z","shell.execute_reply.started":"2022-08-07T15:10:04.972670Z","shell.execute_reply":"2022-08-07T15:13:06.623353Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# ViZ","metadata":{}},{"cell_type":"code","source":"#plot between predicted values and residuals\nplt.scatter(train_f1, train_f1 - train_y, c = \"blue\",  label = \"Training data\")\nplt.scatter(test_f1,test_f1 - val_y, c = \"black\",  label = \"Validation data\")\nplt.title(\"Linear regression\")\nplt.xlabel(\"Predicted values\")\nplt.ylabel(\"Residuals\")\nplt.legend(loc = \"upper left\")\nplt.hlines(y = 0, xmin = 10.5, xmax = 13.5, color = \"red\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-07T15:13:06.626400Z","iopub.execute_input":"2022-08-07T15:13:06.626862Z","iopub.status.idle":"2022-08-07T15:13:06.939711Z","shell.execute_reply.started":"2022-08-07T15:13:06.626815Z","shell.execute_reply":"2022-08-07T15:13:06.938437Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#plot between predicted values and residuals\nplt.scatter(train_f1, train_y, c = \"blue\",  label = \"Training data\")\nplt.scatter(test_f1,val_y, c = \"black\",  label = \"Validation data\")\nplt.title(\"Linear regression\")\nplt.xlabel(\"Predicted values\")\nplt.ylabel(\"Residuals\")\nplt.legend(loc = \"upper left\")\nplt.hlines(y = 0, xmin = 10.5, xmax = 13.5, color = \"red\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-07T15:13:06.941125Z","iopub.execute_input":"2022-08-07T15:13:06.941598Z","iopub.status.idle":"2022-08-07T15:13:07.266960Z","shell.execute_reply.started":"2022-08-07T15:13:06.941564Z","shell.execute_reply":"2022-08-07T15:13:07.265711Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Prediction on Test Set**","metadata":{}},{"cell_type":"code","source":"test_X = df_test[features]\ntest_preds = model_xgb.predict(test_X)\n\nprint(\"Test Predictions: \", test_preds * 100)","metadata":{"execution":{"iopub.status.busy":"2022-08-07T15:13:07.268515Z","iopub.execute_input":"2022-08-07T15:13:07.268943Z","iopub.status.idle":"2022-08-07T15:13:07.284769Z","shell.execute_reply.started":"2022-08-07T15:13:07.268906Z","shell.execute_reply":"2022-08-07T15:13:07.282211Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"output = pd.DataFrame({'Id': df_test.Id,\n                      'SalePrice': test_preds})\n\noutput","metadata":{"execution":{"iopub.status.busy":"2022-08-07T15:13:07.286270Z","iopub.execute_input":"2022-08-07T15:13:07.286638Z","iopub.status.idle":"2022-08-07T15:13:07.301571Z","shell.execute_reply.started":"2022-08-07T15:13:07.286601Z","shell.execute_reply":"2022-08-07T15:13:07.300348Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"output.to_csv('submission2.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-08-07T15:13:07.303192Z","iopub.execute_input":"2022-08-07T15:13:07.303906Z","iopub.status.idle":"2022-08-07T15:13:07.312598Z","shell.execute_reply.started":"2022-08-07T15:13:07.303870Z","shell.execute_reply":"2022-08-07T15:13:07.311336Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}