{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport re # pattern searching\nimport seaborn as sns # statistical plots\nimport matplotlib.pyplot as plt # plots\nfrom scipy import stats\nfrom scipy.stats import norm\nimport plotly.graph_objects as go\nimport plotly.express as px\n%matplotlib inline","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-11T03:35:19.374692Z","iopub.execute_input":"2022-07-11T03:35:19.375315Z","iopub.status.idle":"2022-07-11T03:35:19.38487Z","shell.execute_reply.started":"2022-07-11T03:35:19.375265Z","shell.execute_reply":"2022-07-11T03:35:19.384033Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Loading the training and testing data**","metadata":{}},{"cell_type":"code","source":"df_train = pd.read_csv('../input/house-prices-advanced-regression-techniques/train.csv', index_col='Id')\ndf_test = pd.read_csv('../input/house-prices-advanced-regression-techniques/test.csv', index_col='Id')","metadata":{"execution":{"iopub.status.busy":"2022-07-11T03:35:19.634678Z","iopub.execute_input":"2022-07-11T03:35:19.634975Z","iopub.status.idle":"2022-07-11T03:35:19.688035Z","shell.execute_reply.started":"2022-07-11T03:35:19.634941Z","shell.execute_reply":"2022-07-11T03:35:19.686837Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Finding the outliers**","metadata":{}},{"cell_type":"code","source":"is_outlier = (df_train['GrLivArea']>4000)&(df_train['SalePrice']<700000)\n\nfig, ax = plt.subplots(figsize=(12,8))\nsns.scatterplot(data=df_train, x='GrLivArea', y='SalePrice',hue=is_outlier, ax=ax, legend = False);\n\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-11T03:35:19.933698Z","iopub.execute_input":"2022-07-11T03:35:19.933988Z","iopub.status.idle":"2022-07-11T03:35:20.243687Z","shell.execute_reply.started":"2022-07-11T03:35:19.933955Z","shell.execute_reply":"2022-07-11T03:35:20.24278Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<span style=\"font-size:18px;\"> \n    After plotting few critical parameters such as gross living area against house sales price it is clear that there are some outliers which can be really difficult to accommodate in any prediction models. The best way to deal with them is to remove them from the dataset for the sake of modeling\n</span>","metadata":{}},{"cell_type":"code","source":"df_train = df_train.drop(df_train[(df_train['GrLivArea']>4000)&(df_train['SalePrice']<700000)].index)\nfig, ax = plt.subplots(figsize=(12,8))\nsns.scatterplot(data=df_train, x='GrLivArea', y='SalePrice', ax=ax);","metadata":{"execution":{"iopub.status.busy":"2022-07-11T03:35:20.245599Z","iopub.execute_input":"2022-07-11T03:35:20.246309Z","iopub.status.idle":"2022-07-11T03:35:20.498348Z","shell.execute_reply.started":"2022-07-11T03:35:20.24627Z","shell.execute_reply":"2022-07-11T03:35:20.497519Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<span style=\"font-size:18px;\">\n    After removing the outliers, train data (except sales price) and test data are appended them to make a total feature dataset for preprocessing, preparing it for modeling\n</span>","metadata":{}},{"cell_type":"code","source":"train_copy = df_train.copy().drop('SalePrice', axis=1)\ntest_copy = df_test.copy()\ndf_total = pd.concat([train_copy, test_copy])\ndf_total.to_csv('df_total.csv')\ndf_total.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-11T03:35:20.499805Z","iopub.execute_input":"2022-07-11T03:35:20.500015Z","iopub.status.idle":"2022-07-11T03:35:20.646238Z","shell.execute_reply.started":"2022-07-11T03:35:20.499988Z","shell.execute_reply":"2022-07-11T03:35:20.64537Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Preprocessing Techniques**\n<ol style=\"font-size:18px; line-height:200%\">\n    <li>Converting Dtypes</li>\n    <li>Imputation</li>\n    <li>Feature Encoding</li>\n    <li>Removing Highly Correlated Input Features</li>\n    <li>Removing Output Feature Skewness</li>\n    <li>One Hot Encoding</li>\n","metadata":{}},{"cell_type":"markdown","source":"## **1. Converting dtypes**\n<span style=\"font-size:18px;\">\n    Some features have incorrect datatypes and lot of features have missing values. Feature set needs dtype conversion and imputation before it can be used for modelling\n</span>","metadata":{}},{"cell_type":"code","source":"##converting the numerical columns into object columns\ndf_total['GarageYrBlt'] = df_total['GarageYrBlt'].fillna(0).astype('int64')\ndf_total['MSSubClass'] = df_total['MSSubClass'].astype('object')\ndf_total['MoSold'] = df_total['MoSold'].astype('object', copy=False)\ndf_total['YrSold'] = df_total['YrSold'].astype('object', copy=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-11T03:35:20.795115Z","iopub.execute_input":"2022-07-11T03:35:20.795378Z","iopub.status.idle":"2022-07-11T03:35:20.803893Z","shell.execute_reply.started":"2022-07-11T03:35:20.79535Z","shell.execute_reply":"2022-07-11T03:35:20.803227Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## **2. Imputation**","metadata":{}},{"cell_type":"markdown","source":"### **I. Primary Imputation**\n<p style=\"font-size:18px\">\n    In the first stage we will use logical assumptions to impute the missing data.\n</p>\n<ol style=\"font-size:18px; line-height:200%\">\n    <li>MS Zoning: It can be fairly assume that Houses in the same residential zone will possibly have same housing subclass and neighborhood. Based MSSubClass and Neighborhood of the record (with missing MSZoning value), the missing values will be replaced with/imputed as mode value for that combination of MSSubClass and Neighborhood e.g. For a record with missing MSZoning if MSSubClass and Neighborhood values are 150 and 'North Ames' respectively, the mode MSZoning value for all the records with the same pair of MSSubClass and Neighborhood values, Residential Low Density in this case, will be used for imputation.</li>\n    <li>LotFrontage: Similar strategy can be used for LotFrontage as well. For missing LotFrontage, median LotFrontage of all records with same MSZoning value will be imputed</li>\n    <li>Masonry Veneer Area and Type: No logical assumption can be leveraged for this feature. A simply imputation of missing Masonry Veneer Area value with 0 and missing Masonry Veneer Type value with None will be used here</li>\n    <li>Basement features: There are lot of basement features missing where the basement area is zero hence imputing these missing features as 'NA' (categorical) or '0'(numerical) will be the right approach</li>\n    <li>Garage features: There are lot of basement features missing where the garage area is zero and hence imputing these missing features as 'NA' (categorical) or '0'(numerical) will be the right approach</li>\n    <li>Utilities: All but one value is missing. And majority of the values are 'AllPub' (2916 out of 2919).'AllPub' will be used inplace of missing values</li>\n    <li>Exterior features (1st and 2nd): Mode Exterior features corresponding to same MSSubClass and Neighborhood will be used for imputation</li>\n    <li>Electrical, Functional, Kitchen features: Absolute mode values to impute the missing values seems the right way</li>  \n    <li>Fireplaces, Pool and Miscellenious features: Where fireplace/pool area/misc feature value is zero, the corresponding categorical features will be imputed as 'NA'. There are records where pool area is not zero but pool quality is missing. In that case mode pool quality value will be used. There are records where miscellenious feature value is non zero but the miscellenious feature value is missing. In that case we the 'Othr' value will be used for imputation</li>\n    <li>SaleType: Absolute mode value will be used to impute the missing values</li>  \n    <li>Alley and Fence: Most of the records have missing values for these two feature. These columns will be discarded from further preprocessing</li>\n\n</ol>","metadata":{}},{"cell_type":"code","source":"##MSZoning: Mode MSZoning values sharing same MSSubclass and Neighborhood \ndf_total['MSZoning']=df_total.groupby(['MSSubClass', 'Neighborhood'])['MSZoning'].transform(lambda x:x.fillna(x.mode()[0]))\n\n##Lot Features\ndf_total['LotFrontage']=df_total.groupby(\"MSZoning\")['LotFrontage'].transform(lambda x:x.fillna(x.median()))\n\n##Masonry Veneer Features\ndf_total['MasVnrArea'].where(df_total['MasVnrArea'].notna(), 0, inplace=True)\ndf_total['MasVnrType'].where(df_total['MasVnrType'].notna(), \"None\", inplace=True)\n\n\n##Basement Feautres\ndf_total['BsmtQual'].where(df_total['TotalBsmtSF']!=0, \"NA\", inplace=True)\ndf_total['BsmtCond'].where(df_total['TotalBsmtSF']!=0, \"NA\", inplace=True)\ndf_total['BsmtExposure'].where(df_total['TotalBsmtSF']!=0, \"NA\", inplace=True)\ndf_total['BsmtFinType1'].where(df_total['TotalBsmtSF']!=0, \"NA\", inplace=True)\ndf_total['BsmtFinType2'].where(df_total['TotalBsmtSF']!=0, \"NA\", inplace=True)\ndf_total['BsmtFullBath'].where(df_total['TotalBsmtSF']!=0, 0.0, inplace=True)\ndf_total['BsmtHalfBath'].where(df_total['TotalBsmtSF']!=0, 0.0, inplace=True)\ndf_total['BsmtFinSF1'].where(df_total['TotalBsmtSF']!=0, 0.0, inplace=True)\ndf_total['BsmtFinSF2'].where(df_total['TotalBsmtSF']!=0, 0.0, inplace=True)\ndf_total['BsmtUnfSF'].where(df_total['TotalBsmtSF']!=0, 0.0, inplace=True)\n\n\n##Garage Features\ndf_total['GarageCond'].where(df_total['GarageArea']!=0, \"NA\", inplace=True)\ndf_total['GarageQual'].where(df_total['GarageArea']!=0, \"NA\", inplace=True)\ndf_total['GarageFinish'].where(df_total['GarageArea']!=0, \"NA\", inplace=True)\ndf_total['GarageType'].where(df_total['GarageArea']!=0, \"NA\", inplace=True)\n\n\n##Utilities: 2916 out of 2917 realized utilities values are 'AllPub'. We will just use the mode to fill NA values\ndf_total['Utilities']=df_total['Utilities'].fillna(df_total['Utilities'].mode()[0])\n\n\n##Exterior1st and Exterior2nd: Mode Exterior1st and Exterior2nd values sharing same MSSubclass and Neighborhood \ndf_total['Exterior1st']=df_total.groupby(['MSSubClass', 'Neighborhood'])['Exterior1st'].transform(lambda x:x.fillna(x.mode()[0]))\ndf_total['Exterior2nd']=df_total.groupby(['MSSubClass', 'Neighborhood'])['Exterior2nd'].transform(lambda x:x.fillna(x.mode()[0]))\n\n\n##Electrical\ndf_total['Electrical']=df_total['Electrical'].fillna(df_total['Electrical'].mode()[0])\n\n\n##KitchenQual\ndf_total['KitchenQual']=df_total['KitchenQual'].fillna(df_total['KitchenQual'].mode()[0])\n\n\n##Functional\ndf_total['Functional']=df_total['Functional'].fillna(df_total['Functional'].mode()[0])\n\n\n##Fireplace Features\ndf_total['FireplaceQu'].where(df_total['Fireplaces']!=0, \"NA\", inplace=True)\n\n\n##Pool Features\ndf_total['PoolQC'].where(df_total['PoolArea']!=0, \"NA\", inplace=True)\ndf_total['PoolQC']=df_total['PoolQC'].fillna(df_total['PoolQC'][df_total['PoolArea']!=0].mode()[0])\n\n\n##Miscellaneous Features\ndf_total['MiscFeature'].where(df_total['MiscVal']!=0, \"NA\", inplace=True)\ndf_total['MiscFeature']=df_total['MiscFeature'].fillna('Othr')\n\n\n##SaleType\ndf_total['SaleType']=df_total['SaleType'].fillna(df_total['SaleType'].mode()[0])\n\n\n##Too many missing values in Alley and Fence feature and no logical step to impute. We will drop them. \n##Using errors='ignore' will suppress errors if non-existing rows are dropped\ndf_total.drop(axis=1, columns=['Alley', 'Fence'], errors='ignore', inplace=True)\n\n\n##Missing values after primary imputation\ndf_total.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-11T03:35:21.137321Z","iopub.execute_input":"2022-07-11T03:35:21.13764Z","iopub.status.idle":"2022-07-11T03:35:21.455778Z","shell.execute_reply.started":"2022-07-11T03:35:21.137605Z","shell.execute_reply":"2022-07-11T03:35:21.454867Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### **II. Secondary Imputation**","metadata":{}},{"cell_type":"markdown","source":"#### **Basement Features**\n<span style=\"font-size:18px\">\n    There are records where either all of the basement features are missing or the basement area is non zero but corresponding categorical features are missing in that case. In former case we just assume that house does not have any basement and impute numerical values as '0' and categorical values as 'NA' and in latter case we just impute absolute mode value for all the missing features.\n</span>","metadata":{}},{"cell_type":"code","source":"##Id 2121 has no data for all the basement features. We will assume that it does not has any basement\ndf_total.iloc[2120,[28,29,30,31,33]] = 'NA'\ndf_total.iloc[2120,[32,34,35,36,45,46]] = 0.0\n\ndf_total.loc[2121, ['BsmtFinType1',\n 'BsmtFinSF1',\n 'BsmtFinSF2',\n 'BsmtUnfSF',\n 'TotalBsmtSF',\n 'BsmtFullBath',\n 'BsmtHalfBath']]=0.0\n\n##BsmtQual: Mode BsmtQual values for NaN\ndf_total['BsmtQual']=df_total['BsmtQual'].transform(lambda x: x.fillna(x.mode()[0]))\n\n##BsmtCond: Mode BsmtCond values for NaN\ndf_total['BsmtCond']=df_total['BsmtCond'].transform(lambda x: x.fillna(x.mode()[0]))\n\n##BsmtExposure: Mode BsmtExposure values for NaN\ndf_total['BsmtExposure']=df_total['BsmtExposure'].transform(lambda x: x.fillna(x.mode()[0]))\n\n##BsmtFinTyoe2: Mode BsmtFinType2 values (that have non zero BsmtFinSF2 which is less than 500) for NaN values\ndf_total['BsmtFinType2']=df_total['BsmtFinType2'].transform(lambda x: x.fillna('Rec'))\n\ndf_total.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-11T03:35:21.457701Z","iopub.execute_input":"2022-07-11T03:35:21.45825Z","iopub.status.idle":"2022-07-11T03:35:21.504222Z","shell.execute_reply.started":"2022-07-11T03:35:21.458212Z","shell.execute_reply":"2022-07-11T03:35:21.503615Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### **Garage Features**\n\n<span style=\"font-size:18px\">\n    There are two records where garage features are missing. In one case the garage area is non zero and in other case garage area is not available. Both of these garages are detached type. Using this as a common feature we find mode values to impute missing garage quality, finish, condition and cars. After we find the garage cars value for the house where area is missing we can impute the mean garage area where type is detached and garage cars value is same as in case of missing garage area record.\n</span>","metadata":{"execution":{"iopub.status.busy":"2022-03-09T05:30:38.171052Z","iopub.execute_input":"2022-03-09T05:30:38.171354Z","iopub.status.idle":"2022-03-09T05:30:38.195612Z","shell.execute_reply.started":"2022-03-09T05:30:38.171322Z","shell.execute_reply":"2022-03-09T05:30:38.194329Z"}}},{"cell_type":"code","source":"##GarageFinish\ndf_total['GarageFinish']=df_total['GarageFinish'].fillna(df_total['GarageFinish'][df_total['GarageType']=='Detchd'].mode()[0])\n\n##GarageQual\ndf_total['GarageQual']=df_total['GarageQual'].fillna(df_total['GarageQual'][df_total['GarageType']=='Detchd'].mode()[0])\n\n##GarageCond\ndf_total['GarageCond']=df_total['GarageCond'].fillna(df_total['GarageCond'][df_total['GarageType']=='Detchd'].mode()[0])\n\n##GarageCars\ndf_total['GarageCars']=df_total['GarageCars'].fillna(df_total['GarageCars'][df_total['GarageType']=='Detchd'].mode()[0])\n\n##GarageArea\ndf_total['GarageArea']=df_total['GarageArea'].fillna(df_total['GarageArea'][(df_total['GarageType']=='Detchd')&(df_total['GarageCars']==2.0)].mean())\n\ndf_total.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-11T03:35:21.709459Z","iopub.execute_input":"2022-07-11T03:35:21.709791Z","iopub.status.idle":"2022-07-11T03:35:21.756574Z","shell.execute_reply.started":"2022-07-11T03:35:21.709755Z","shell.execute_reply":"2022-07-11T03:35:21.755646Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## **3. Encoding the ordinal features into numerical features**\n\n<span style=\"font-size:18px\">\n    To process the ordinal features in the models we need to convert them into numerical features first. We will assume one step integer increment for each ordinal category.\n</span>","metadata":{}},{"cell_type":"code","source":"df_total = df_total.replace({\n    \"BsmtCond\" : {\"NA\" : 0, \"Po\" : 1, \"Fa\" : 2, \"TA\" : 3, \"Gd\" : 4, \"Ex\" : 5},\n    \"BsmtExposure\" : {\"NA\" : 0,\"No\":1, \"Mn\" : 2, \"Av\": 3, \"Gd\" : 4},\n    \"BsmtFinType1\" : {\"NA\" : 0, \"Unf\" : 1, \"LwQ\": 2, \"Rec\" : 3, \"BLQ\" : 4, \"ALQ\" : 5, \"GLQ\" : 6},\n    \"BsmtFinType2\" : {\"NA\" : 0, \"Unf\" : 1, \"LwQ\": 2, \"Rec\" : 3, \"BLQ\" : 4, \"ALQ\" : 5, \"GLQ\" : 6},\n    \"BsmtQual\" : {\"NA\" : 0, \"Po\" : 1, \"Fa\" : 2, \"TA\": 3, \"Gd\" : 4, \"Ex\" : 5},\n    \"ExterCond\" : {\"Po\" : 1, \"Fa\" : 2, \"TA\": 3, \"Gd\": 4, \"Ex\" : 5},\n    \"ExterQual\" : {\"Po\" : 1, \"Fa\" : 2, \"TA\": 3, \"Gd\": 4, \"Ex\" : 5},\n    \"FireplaceQu\" : {\"NA\" : 0, \"Po\" : 1, \"Fa\" : 2, \"TA\" : 3, \"Gd\" : 4, \"Ex\" : 5},\n    \"Functional\" : {\"Sal\" : 1, \"Sev\" : 2, \"Maj2\" : 3, \"Maj1\" : 4, \"Mod\": 5, \"Min2\" : 6, \"Min1\" : 7, \"Typ\" : 8},\n    \"GarageFinish\" : {\"NA\" : 0, \"Unf\" : 1, \"RFn\" : 2, \"Fin\" : 3},\n    \"GarageCond\" : {\"NA\" : 0, \"Po\" : 1, \"Fa\" : 2, \"TA\" : 3, \"Gd\" : 4, \"Ex\" : 5},\n    \"GarageQual\" : {\"NA\" : 0, \"Po\" : 1, \"Fa\" : 2, \"TA\" : 3, \"Gd\" : 4, \"Ex\" : 5},\n    \"HeatingQC\" : {\"Po\" : 1, \"Fa\" : 2, \"TA\" : 3, \"Gd\" : 4, \"Ex\" : 5},\n    \"KitchenQual\" : {\"Po\" : 1, \"Fa\" : 2, \"TA\" : 3, \"Gd\" : 4, \"Ex\" : 5},\n    \"PavedDrive\" : {\"NA\" : 0, \"P\" : 1, \"Y\" : 2},\n    \"PoolQC\" : {\"NA\" : 0, \"Fa\" : 1, \"TA\" : 2, \"Gd\" : 3, \"Ex\" : 4},\n    \"CentralAir\":{\"N\":0,\"Y\":1}})","metadata":{"execution":{"iopub.status.busy":"2022-07-11T03:35:21.990669Z","iopub.execute_input":"2022-07-11T03:35:21.990941Z","iopub.status.idle":"2022-07-11T03:35:22.077527Z","shell.execute_reply.started":"2022-07-11T03:35:21.990912Z","shell.execute_reply":"2022-07-11T03:35:22.076349Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"\n<span style=\"font-size:18px\">\n    For correlation and skewness measurement let us create seperate dataframe for numerical and categorical features\n</span>","metadata":{}},{"cell_type":"code","source":"df_num_ord = df_total.select_dtypes(include=['int64','float64'])\ndf_cat = df_total.select_dtypes(include=['object']).astype(str)","metadata":{"execution":{"iopub.status.busy":"2022-07-11T03:35:22.244182Z","iopub.execute_input":"2022-07-11T03:35:22.244468Z","iopub.status.idle":"2022-07-11T03:35:22.259947Z","shell.execute_reply.started":"2022-07-11T03:35:22.244436Z","shell.execute_reply":"2022-07-11T03:35:22.259271Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## **4. Removing Highly Correlated Numerical Columns**\n<span style=\"font-size:18px\">\n    The input features to be used in the regression model should be atleast fairly uncorrelated if not completed independent. Correlation matrix visualizes the correlation coefficient between features. Since some of the numerical features were originally ordinal, spearman correlation coefficient performs better than pearson correlation coefficient. Spearman coefficient captures the correlation even when it is fairly qualitative.\n</span>","metadata":{}},{"cell_type":"code","source":"corr = df_num_ord.corr(method='spearman')\nkot = corr[((corr>=.7) | (corr<=-0.7)) & (corr!=1.0)]\nfig, ax = plt.subplots(figsize=(18,18))\nsns.heatmap(kot,ax=ax,center=0,annot=True,fmt='0.2f',cbar=False,cmap=sns.diverging_palette(110,110,s=100,l=60,center='light',as_cmap=True));\nfig.savefig('num_corr.jpeg', dpi=1200, bbox_inches='tight')","metadata":{"execution":{"iopub.status.busy":"2022-07-11T03:35:22.513274Z","iopub.execute_input":"2022-07-11T03:35:22.513767Z","iopub.status.idle":"2022-07-11T03:35:44.628072Z","shell.execute_reply.started":"2022-07-11T03:35:22.513722Z","shell.execute_reply":"2022-07-11T03:35:44.627035Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<span style=\"font-size:18px\">\n    Highly correlated (positively or negatively) features will generate a model parameters having high variance. Hence they need to be eliminated.\n</span>","metadata":{}},{"cell_type":"code","source":"def correlation(dataset, threshold):\n    col_corr = set()\n#     row_corr = set()\n    # Set of all the names of correlated columns\n    corr_matrix = dataset.corr(method='spearman')\n    for i in range(len(corr_matrix.columns)):\n        for j in range(i):\n            if abs(corr_matrix.iloc[i, j]) > threshold: # we are interested in absolute coeff value\n                colname = corr_matrix.columns[i]  # getting the name of column\n#                 rowname = corr_matrix.columns[j]\n                col_corr.add(colname)\n#                 row_corr.add(rowname)\n    return col_corr","metadata":{"execution":{"iopub.status.busy":"2022-07-11T03:35:44.62995Z","iopub.execute_input":"2022-07-11T03:35:44.630689Z","iopub.status.idle":"2022-07-11T03:35:44.637993Z","shell.execute_reply.started":"2022-07-11T03:35:44.630644Z","shell.execute_reply":"2022-07-11T03:35:44.637425Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<span style=\"font-size:18px\">\n    Features with correlation coefficient greater than 0.7 are excluded from the feature set to be used for modeling\n</span>","metadata":{}},{"cell_type":"code","source":"high_corr_features = correlation(df_num_ord, 0.7)\ndf_num_ord.drop(high_corr_features, axis=1, inplace=True, errors='ignore')\ndf_total.drop(high_corr_features, axis=1, inplace=True, errors='ignore')","metadata":{"execution":{"iopub.status.busy":"2022-07-11T03:35:44.639287Z","iopub.execute_input":"2022-07-11T03:35:44.640244Z","iopub.status.idle":"2022-07-11T03:35:44.727172Z","shell.execute_reply.started":"2022-07-11T03:35:44.640198Z","shell.execute_reply":"2022-07-11T03:35:44.726368Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## **5. Removing Output Feature Skewness**","metadata":{}},{"cell_type":"code","source":"sns.distplot(df_train['SalePrice'] , fit=norm);\n\n# Get the fitted parameters used by the function\n(mu, sigma) = norm.fit(df_train['SalePrice'])\nprint( '\\n mu = {:.2f} and sigma = {:.2f}\\n'.format(mu, sigma))\n\n#Now plot the distribution\nplt.legend(['Normal dist. ($\\mu=$ {:.2f} and $\\sigma=$ {:.2f} )'.format(mu, sigma)],\n            loc='best')\nplt.ylabel('Frequency')\nplt.title('SalePrice distribution')\n\n#Get also the QQ-plot\nfig = plt.figure()\nres = stats.probplot(df_train['SalePrice'], plot=plt)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-11T03:35:44.731045Z","iopub.execute_input":"2022-07-11T03:35:44.731414Z","iopub.status.idle":"2022-07-11T03:35:45.312594Z","shell.execute_reply.started":"2022-07-11T03:35:44.731384Z","shell.execute_reply":"2022-07-11T03:35:45.311879Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<span style=\"font-size:18px\">\n    The Sales Price distribution is skewed with heavy right tail. Since one of the common assumption of Linear Regression is Normal Distribution of errors (which translates to normal distribution of output variable given input variables and normal distribution of the errors), the output variable should to Normally/near-Normally Distributed. Log1p transform can be applied to remove the skewness.\n</span>","metadata":{}},{"cell_type":"code","source":"#We use the numpy fuction log1p which  applies log(1+x) to all elements of the column\ndf_train[\"SalePrice\"] = np.log1p(df_train[\"SalePrice\"])\n\n#Check the new distribution \nsns.distplot(df_train['SalePrice'] , fit=norm);\n\n# Get the fitted parameters used by the function\n(mu, sigma) = norm.fit(df_train['SalePrice'])\nprint( '\\n mu = {:.2f} and sigma = {:.2f}\\n'.format(mu, sigma))\n\n#Now plot the distribution\nplt.legend(['Normal dist. ($\\mu=$ {:.2f} and $\\sigma=$ {:.2f} )'.format(mu, sigma)],\n            loc='best')\nplt.ylabel('Frequency')\nplt.title('SalePrice distribution')\n\n#Get also the QQ-plot\nfig = plt.figure()\nres = stats.probplot(df_train['SalePrice'], plot=plt)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-11T03:35:45.314047Z","iopub.execute_input":"2022-07-11T03:35:45.314506Z","iopub.status.idle":"2022-07-11T03:35:45.87393Z","shell.execute_reply.started":"2022-07-11T03:35:45.31446Z","shell.execute_reply":"2022-07-11T03:35:45.872976Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<span style=\"font-size:18px\">\n    Applying log transform made the sales price near normally distributed.\n</span>","metadata":{}},{"cell_type":"markdown","source":"## **6. One hot encoding**\n<span style=\"font-size:18px\">\n    Finally we need to perform one hot encoding of the categorical variable. To avoid multicollinearity among the input variables we will drop the first column of every one hot encoded variable.\n</span>","metadata":{}},{"cell_type":"code","source":"df_total = pd.get_dummies(df_total,drop_first=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-11T03:35:45.875497Z","iopub.execute_input":"2022-07-11T03:35:45.875987Z","iopub.status.idle":"2022-07-11T03:35:45.921439Z","shell.execute_reply.started":"2022-07-11T03:35:45.875952Z","shell.execute_reply":"2022-07-11T03:35:45.92072Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Modeling Techniques**\n<ol style=\"font-size:18px; line-height:200%\">\n    <li>Linear Models</li>\n    <li>Decision Tree Regressions</li>\n    <li>Regularized Liner Models</li>\n    <li>Grid Search for Parameter Optimization</li>\n    <li>Stacking Best Regressors</li>\n    <li>Extreme Gradient Boosting</li>\n    <li>Averaging Best Outputs</li>","metadata":{}},{"cell_type":"code","source":"from sklearn.linear_model import LinearRegression\nfrom sklearn.ensemble import RandomForestRegressor\nfrom sklearn.linear_model import Ridge\nfrom sklearn.linear_model import Lasso\nfrom sklearn.linear_model import ElasticNet\nfrom sklearn.model_selection import GridSearchCV\nfrom sklearn.ensemble import StackingRegressor\nfrom xgboost import XGBRegressor\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import r2_score\nfrom mpl_toolkits.mplot3d import Axes3D","metadata":{"execution":{"iopub.status.busy":"2022-07-11T03:35:45.922752Z","iopub.execute_input":"2022-07-11T03:35:45.922974Z","iopub.status.idle":"2022-07-11T03:35:45.930614Z","shell.execute_reply.started":"2022-07-11T03:35:45.922947Z","shell.execute_reply":"2022-07-11T03:35:45.929716Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#records and features in the training dataset\nn_train = df_train.index[-1] \n\nX=df_total.loc[:n_train]\ny=df_train['SalePrice']\n\nX_train, X_test, y_train, y_test = train_test_split(X,y,test_size=0.2)","metadata":{"execution":{"iopub.status.busy":"2022-07-11T03:35:45.931507Z","iopub.execute_input":"2022-07-11T03:35:45.932431Z","iopub.status.idle":"2022-07-11T03:35:45.954881Z","shell.execute_reply.started":"2022-07-11T03:35:45.93238Z","shell.execute_reply":"2022-07-11T03:35:45.954087Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## **1. Linear Models (Linear Regression)**","metadata":{}},{"cell_type":"code","source":"LR = LinearRegression()\nLR.fit(X_train, y_train)\ny_pred=LR.predict(X_test)\n\nr2 = r2_score(y_test, y_pred)\nrmse = np.sqrt(np.mean((y_test-y_pred)**2))\nysub_pred_LR = np.expm1(LR.predict(df_total.loc[n_train+1:]))\nsubmission = pd.DataFrame(data=ysub_pred_LR,index=df_test.index,columns=['SalePrice'])\nsubmission.to_csv('LR_final.csv')\nr2, rmse","metadata":{"execution":{"iopub.status.busy":"2022-07-11T03:35:45.956283Z","iopub.execute_input":"2022-07-11T03:35:45.957248Z","iopub.status.idle":"2022-07-11T03:35:46.104489Z","shell.execute_reply.started":"2022-07-11T03:35:45.957195Z","shell.execute_reply":"2022-07-11T03:35:46.102672Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<span style=\"font-size:18px\">\n    Residual vs output variable plot determines the extent of homoscedasticity of the residuals. The above plot shows that residuals are not perfectly homoscedastic although the range of residual variance is quite narrow. We will explore other regression models as well and check its homoscedasticity.\n</span>","metadata":{}},{"cell_type":"code","source":"residual = y_test - y_pred\nplt.scatter(y_pred, residual)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-11T03:35:46.11464Z","iopub.execute_input":"2022-07-11T03:35:46.1151Z","iopub.status.idle":"2022-07-11T03:35:46.284369Z","shell.execute_reply.started":"2022-07-11T03:35:46.11505Z","shell.execute_reply":"2022-07-11T03:35:46.283503Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## **2. Decision Tree Regression (Random Forest Regression)**","metadata":{}},{"cell_type":"code","source":"RF = RandomForestRegressor(n_estimators=100, max_depth=12)\nRF.fit(X_train, y_train)\ny_pred=RF.predict(X_test)\n\nr2 = r2_score(y_test, y_pred)\nrmse = np.sqrt(np.mean((y_test-y_pred)**2))\nysub_pred_RF = np.expm1(RF.predict(df_total.loc[n_train+1:]))\nsubmission = pd.DataFrame(data=ysub_pred_RF,index=df_test.index,columns=['SalePrice'])\nsubmission.to_csv('RF_final.csv')\nr2, rmse","metadata":{"execution":{"iopub.status.busy":"2022-07-11T03:35:46.286165Z","iopub.execute_input":"2022-07-11T03:35:46.286543Z","iopub.status.idle":"2022-07-11T03:35:47.990487Z","shell.execute_reply.started":"2022-07-11T03:35:46.286455Z","shell.execute_reply":"2022-07-11T03:35:47.989612Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"residual = y_test - y_pred\nplt.scatter(y_pred, residual)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-11T03:35:47.991758Z","iopub.execute_input":"2022-07-11T03:35:47.992027Z","iopub.status.idle":"2022-07-11T03:35:48.123306Z","shell.execute_reply.started":"2022-07-11T03:35:47.991998Z","shell.execute_reply":"2022-07-11T03:35:48.122637Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## **3. Regularized Linear Models**","metadata":{"execution":{"iopub.status.busy":"2022-03-19T21:35:26.18418Z","iopub.execute_input":"2022-03-19T21:35:26.184451Z","iopub.status.idle":"2022-03-19T21:35:26.198266Z","shell.execute_reply.started":"2022-03-19T21:35:26.184422Z","shell.execute_reply":"2022-03-19T21:35:26.197203Z"}}},{"cell_type":"markdown","source":"### **I. Ridge Regression**","metadata":{}},{"cell_type":"code","source":"Rdge = Ridge(alpha=10)\nRdge.fit(X_train, y_train)\ny_pred=Rdge.predict(X_test)\n\nr2 = r2_score(y_test, y_pred)\nrmse = np.sqrt(np.mean((y_test - y_pred)**2))\nysub_pred_Ridge = np.expm1(Rdge.predict(df_total.loc[n_train+1:]))\nsubmission = pd.DataFrame(data=ysub_pred_Ridge,index=df_test.index,columns=['SalePrice'])\nsubmission.to_csv('Rdge_final.csv')\nr2, rmse","metadata":{"execution":{"iopub.status.busy":"2022-07-11T03:35:48.124492Z","iopub.execute_input":"2022-07-11T03:35:48.124873Z","iopub.status.idle":"2022-07-11T03:35:48.232264Z","shell.execute_reply.started":"2022-07-11T03:35:48.12484Z","shell.execute_reply":"2022-07-11T03:35:48.231067Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"residual = y_test - y_pred\nplt.scatter(y_pred, residual)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-11T03:35:48.235105Z","iopub.execute_input":"2022-07-11T03:35:48.237097Z","iopub.status.idle":"2022-07-11T03:35:48.414436Z","shell.execute_reply.started":"2022-07-11T03:35:48.237036Z","shell.execute_reply":"2022-07-11T03:35:48.413704Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### **II. Lasso Regression**","metadata":{}},{"cell_type":"code","source":"Lsso = Lasso(alpha=0.5)\nLsso.fit(X_train, y_train)\n\nr2 = r2_score(y_test, y_pred)\nrmse = np.sqrt(np.mean((y_test - y_pred)**2))\nysub_pred_Lasso = np.expm1(Lsso.predict(df_total.loc[n_train+1:]))\nsubmission = pd.DataFrame(data=ysub_pred_Lasso,index=df_test.index,columns=['SalePrice'])\nsubmission.to_csv('Lsso_final.csv')\nr2, rmse","metadata":{"execution":{"iopub.status.busy":"2022-07-11T03:35:48.41572Z","iopub.execute_input":"2022-07-11T03:35:48.416121Z","iopub.status.idle":"2022-07-11T03:35:48.505885Z","shell.execute_reply.started":"2022-07-11T03:35:48.41609Z","shell.execute_reply":"2022-07-11T03:35:48.504893Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"residual = y_test - y_pred\nplt.scatter(y_pred, residual)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-11T03:35:48.507483Z","iopub.execute_input":"2022-07-11T03:35:48.50921Z","iopub.status.idle":"2022-07-11T03:35:48.73123Z","shell.execute_reply.started":"2022-07-11T03:35:48.509155Z","shell.execute_reply":"2022-07-11T03:35:48.730285Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### **III. Elasic Net Regression**","metadata":{}},{"cell_type":"code","source":"EN = ElasticNet(alpha=5, l1_ratio=0.05)\nEN.fit(X_train, y_train)\n\nr2 = r2_score(y_test, y_pred)\nrmse = np.sqrt(np.mean((y_test - y_pred)**2))\nysub_pred_EN = np.expm1(EN.predict(df_total.loc[n_train+1:]))\nsubmission = pd.DataFrame(data=ysub_pred_EN,index=df_test.index,columns=['SalePrice'])\nsubmission.to_csv('Lsso_final.csv')\nr2, rmse","metadata":{"execution":{"iopub.status.busy":"2022-07-11T03:35:48.732435Z","iopub.execute_input":"2022-07-11T03:35:48.732656Z","iopub.status.idle":"2022-07-11T03:35:48.787853Z","shell.execute_reply.started":"2022-07-11T03:35:48.732628Z","shell.execute_reply":"2022-07-11T03:35:48.786896Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"residual = y_test - y_pred\nplt.scatter(y_pred, residual)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-11T03:35:48.793625Z","iopub.execute_input":"2022-07-11T03:35:48.7965Z","iopub.status.idle":"2022-07-11T03:35:49.037373Z","shell.execute_reply.started":"2022-07-11T03:35:48.796435Z","shell.execute_reply":"2022-07-11T03:35:49.036524Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## **4. Grid Search for Parameter Optimization**\n<p style=\"font-size:18px;\"> \n    Let us use cross validation, and grid search to find best parameters for ridge, lasso, elasticnet\n</p>\n<p style=\"font-size:18px;\"> \n    Note: The parameter values specified below are values which gave best r2 score for 5-fold cross validation in each case. \n</p>","metadata":{}},{"cell_type":"markdown","source":"### **I. Ridge Regression**","metadata":{}},{"cell_type":"code","source":"parameters = {'alpha' : [1, 5, 8, 10, 12, 13, 14, 15, 16, 18, 20]}\nscoring = ['neg_mean_squared_error', 'r2']\nridgeCV = GridSearchCV(Ridge(), param_grid=parameters, scoring=scoring, refit='r2', cv=5, return_train_score=True).fit(X, y)\nprint(f\"Best Ridge Regressor is {ridgeCV.best_estimator_} and corresponding r2 score on prediction is {ridgeCV.best_score_}\")\n\nysub_pred_ridgeCV = np.expm1(ridgeCV.predict(df_total.loc[n_train+1:]))\nsubmission = pd.DataFrame(data=ysub_pred_ridgeCV,index=df_test.index,columns=['SalePrice'])\nsubmission.to_csv('ridgeCV_final.csv')","metadata":{"execution":{"iopub.status.busy":"2022-07-11T03:35:49.038659Z","iopub.execute_input":"2022-07-11T03:35:49.03909Z","iopub.status.idle":"2022-07-11T03:35:52.424948Z","shell.execute_reply.started":"2022-07-11T03:35:49.039038Z","shell.execute_reply":"2022-07-11T03:35:52.423905Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"params = [i['alpha'] for i  in ridgeCV.cv_results_['params']]\nmean_train_r2 = ridgeCV.cv_results_['mean_train_r2']\nmean_test_r2 = ridgeCV.cv_results_['mean_test_r2']\nfig, ax = plt.subplots(figsize=(12,8))\nsns.lineplot(x=params, y=mean_train_r2, ax=ax)\nsns.lineplot(x=params, y=mean_test_r2, ax=ax)\nplt.legend(['mean_train_r2', 'mean_test_r2'])\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-11T03:35:52.429797Z","iopub.execute_input":"2022-07-11T03:35:52.433774Z","iopub.status.idle":"2022-07-11T03:35:52.739934Z","shell.execute_reply.started":"2022-07-11T03:35:52.43371Z","shell.execute_reply":"2022-07-11T03:35:52.73905Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### **II. Lasso Regression**","metadata":{}},{"cell_type":"code","source":"parameters = {'alpha' : [1e-4, 2.5e-4, 3e-4, 3.5e-4, 4e-4, 4.5e-4, 5e-4, 6e-4, 7e-4]}\nscoring = ['neg_mean_squared_error', 'r2']\nlassoCV = GridSearchCV(Lasso(), param_grid=parameters, scoring=scoring, refit='r2', cv=5, return_train_score=True).fit(X, y)\nprint(f\"Best Lasso Regressor is {lassoCV.best_estimator_},  and corresponding r2 score on prediction is {lassoCV.best_score_}\")\n\nysub_pred_lassoCV = np.expm1(lassoCV.predict(df_total.loc[n_train+1:]))\nsubmission = pd.DataFrame(data=ysub_pred_lassoCV,index=df_test.index,columns=['SalePrice'])\nsubmission.to_csv('lassoCV_final.csv')","metadata":{"execution":{"iopub.status.busy":"2022-07-11T03:35:52.741311Z","iopub.execute_input":"2022-07-11T03:35:52.742184Z","iopub.status.idle":"2022-07-11T03:35:57.987086Z","shell.execute_reply.started":"2022-07-11T03:35:52.742145Z","shell.execute_reply":"2022-07-11T03:35:57.986085Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"params = [i['alpha'] for i  in lassoCV.cv_results_['params']]\nmean_train_r2 = lassoCV.cv_results_['mean_train_r2']\nmean_test_r2 = lassoCV.cv_results_['mean_test_r2']\nfig, ax = plt.subplots(figsize=(12,8))\nax = plt.plot(params, mean_train_r2)\nax = plt.plot(params, mean_test_r2)\nplt.legend(['mean_train_r2', 'mean_test_r2'])\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-11T03:35:57.98884Z","iopub.execute_input":"2022-07-11T03:35:57.990922Z","iopub.status.idle":"2022-07-11T03:35:58.267392Z","shell.execute_reply.started":"2022-07-11T03:35:57.99087Z","shell.execute_reply":"2022-07-11T03:35:58.266511Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### **III. Elastic Net Regression**","metadata":{}},{"cell_type":"code","source":"parameters = {'alpha' : [0.005, 0.007, 0.01, 0.03, 0.05], 'l1_ratio' : [1e-5, 3e-5, 5e-5, 7e-5]}\nscoring = ['neg_mean_squared_error', 'r2']\nenetCV = GridSearchCV(ElasticNet(), param_grid=parameters, scoring=scoring, refit='r2', cv=5, return_train_score=True).fit(X, y)\nprint(f\"Best ElasticNet Regressor is {enetCV.best_estimator_}, and corresponding r2 score on prediction is {enetCV.best_score_}\")\n\nysub_pred_enetCV = np.expm1(enetCV.predict(df_total.loc[n_train+1:]))\nsubmission = pd.DataFrame(data=ysub_pred_enetCV,index=df_test.index,columns=['SalePrice'])\nsubmission.to_csv('enetCV_final.csv')","metadata":{"execution":{"iopub.status.busy":"2022-07-11T03:36:01.48223Z","iopub.execute_input":"2022-07-11T03:36:01.482521Z","iopub.status.idle":"2022-07-11T03:36:18.005275Z","shell.execute_reply.started":"2022-07-11T03:36:01.482492Z","shell.execute_reply":"2022-07-11T03:36:18.004285Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"alpha = [i['alpha'] for i  in enetCV.cv_results_['params']]\nl1_ratio = [i['l1_ratio'] for i  in enetCV.cv_results_['params']]\nmean_train_r2 = enetCV.cv_results_['mean_train_r2']\nmean_test_r2 = enetCV.cv_results_['mean_test_r2']\n\nX, Y=np.array(alpha), np.array(l1_ratio)\nZ1=np.array(mean_train_r2)\nZ2=np.array(mean_test_r2)\n\nfig = go.Figure(data=[go.Mesh3d(z=Z1, x=X, y=Y, color='rgb(150,250,250)', opacity=0.4, name='Training r2 Score', showlegend=True)])\nfig.add_trace(go.Mesh3d(z=Z2, x=X, y=Y, color='rgb(150,150,50)', opacity=0.4,  name='Testing r2 Score', showlegend=True))\n#fig.update_traces(color='rgb(200,250,250)')\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-11T03:41:09.623304Z","iopub.execute_input":"2022-07-11T03:41:09.624383Z","iopub.status.idle":"2022-07-11T03:41:09.642357Z","shell.execute_reply.started":"2022-07-11T03:41:09.624328Z","shell.execute_reply":"2022-07-11T03:41:09.641432Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## **5. Stacking regression**","metadata":{}},{"cell_type":"code","source":"estimators = [('ridgeCV', ridgeCV), ('lassoCV', lassoCV)]\nstackCV = StackingRegressor(estimators=estimators, final_estimator=lassoCV)\nstackCV.fit(X_train, y_train)\n\nr2 = r2_score(y_test, y_pred)\nrmse = np.sqrt(np.mean((y_test - y_pred)**2))\nysub_pred = np.expm1(stackCV.predict(df_total.loc[n_train+1:]))\nsubmission = pd.DataFrame(data=ysub_pred,index=df_test.index,columns=['SalePrice'])\nsubmission.to_csv('stackCV_final.csv')\nr2, rmse","metadata":{"execution":{"iopub.status.busy":"2022-07-08T01:50:33.856303Z","iopub.execute_input":"2022-07-08T01:50:33.858155Z","iopub.status.idle":"2022-07-08T01:50:38.000593Z","shell.execute_reply.started":"2022-07-08T01:50:33.858105Z","shell.execute_reply":"2022-07-08T01:50:37.999605Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## **6. Extreme Gradient Boosting**","metadata":{}},{"cell_type":"code","source":"XGB = XGBRegressor(n_estimators=600,\n                   learning_rate=0.05,\n                   max_depth=4,\n                   booster='dart',\n                   objective='reg:squarederror')\nXGB.fit(X_train, y_train)\n\nysub_pred_XGB = np.exp(XGB.predict(df_total.loc[n_train+1:]))-1\nsubmission = pd.DataFrame(data=ysub_pred_XGB, index=df_test.index, columns=['SalePrice'])\nsubmission.to_csv('XGB_final.csv')","metadata":{"execution":{"iopub.status.busy":"2022-07-08T02:04:26.780898Z","iopub.execute_input":"2022-07-08T02:04:26.781557Z","iopub.status.idle":"2022-07-08T02:05:23.672669Z","shell.execute_reply.started":"2022-07-08T02:04:26.781517Z","shell.execute_reply":"2022-07-08T02:05:23.671994Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## **7. Averaging Best Outputs**\n<p style=\"font-size:18px;\"> \n    One common technique to improve the output accuracy is to average out the outputs predicted by some of the best models\n</p>","metadata":{"execution":{"iopub.status.busy":"2022-07-08T02:06:11.914325Z","iopub.execute_input":"2022-07-08T02:06:11.914717Z","iopub.status.idle":"2022-07-08T02:06:13.559431Z","shell.execute_reply.started":"2022-07-08T02:06:11.91468Z","shell.execute_reply":"2022-07-08T02:06:13.558671Z"}}},{"cell_type":"code","source":"ysub_pred_avg = 0.4*ysub_pred_ridgeCV + 0.6*ysub_pred_XGB\nsubmission = pd.DataFrame(data=ysub_pred_avg, index=df_test.index, columns=['SalePrice'])\nsubmission.to_csv('Average_final.csv')","metadata":{"execution":{"iopub.status.busy":"2022-07-08T02:14:22.609158Z","iopub.execute_input":"2022-07-08T02:14:22.609685Z","iopub.status.idle":"2022-07-08T02:14:22.621984Z","shell.execute_reply.started":"2022-07-08T02:14:22.609652Z","shell.execute_reply":"2022-07-08T02:14:22.621253Z"},"trusted":true},"execution_count":null,"outputs":[]}]}