{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-12T08:32:14.048599Z","iopub.execute_input":"2022-08-12T08:32:14.049013Z","iopub.status.idle":"2022-08-12T08:32:14.057557Z","shell.execute_reply.started":"2022-08-12T08:32:14.048975Z","shell.execute_reply":"2022-08-12T08:32:14.056414Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Introduction ","metadata":{}},{"cell_type":"markdown","source":"**Hello Kagglers!**\n\nIn this notebook you will find a basic **flow** I follow when working **on regression tasks**. \n\nAs the title suggests it's just prototyping and still rough around the edges, but as always, data science is a work in progress. \n\nStructure: \n\n* Data Ingesting\n* Quick EDA\n* Data Cleaning\n* Data Preprocessing\n* Modeling - Stacking predictors\n* Submitting ","metadata":{}},{"cell_type":"markdown","source":"#### TODO:\n\n* Experiment on more feature engineering\n* Impute NA's on 8 columns that I dropped and test if it has better results \n* Model interpretability\n","metadata":{}},{"cell_type":"markdown","source":"#### Importing libraries ","metadata":{}},{"cell_type":"code","source":"import pandas as pd \npd.options.mode.chained_assignment = None  # default='warn'\npd.set_option('display.max_rows',None)\nimport numpy as np \nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport math\nfrom scipy.stats import norm, skew, kurtosis\n%matplotlib inline \nfrom sklearn.impute import SimpleImputer\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import OrdinalEncoder,StandardScaler\nfrom sklearn.metrics import mean_squared_error as mse\nfrom sklearn.metrics import r2_score\nfrom sklearn.linear_model import Ridge, HuberRegressor\nfrom sklearn.svm import SVR\nimport catboost as cb\nfrom sklearn.ensemble import StackingRegressor\nfrom category_encoders import MEstimateEncoder\nfrom xgboost import XGBRegressor\n","metadata":{"execution":{"iopub.status.busy":"2022-08-12T08:44:55.168084Z","iopub.execute_input":"2022-08-12T08:44:55.168505Z","iopub.status.idle":"2022-08-12T08:44:55.180766Z","shell.execute_reply.started":"2022-08-12T08:44:55.168470Z","shell.execute_reply":"2022-08-12T08:44:55.179305Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Ingesting data","metadata":{}},{"cell_type":"code","source":"train_df = pd.read_csv(\"/kaggle/input/house-prices-advanced-regression-techniques/train.csv\")\ntest_df = pd.read_csv(\"/kaggle/input/house-prices-advanced-regression-techniques/test.csv\")\nsample_submission = pd.read_csv('/kaggle/input/house-prices-advanced-regression-techniques/sample_submission.csv')","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2022-08-12T08:44:55.970803Z","iopub.execute_input":"2022-08-12T08:44:55.971884Z","iopub.status.idle":"2022-08-12T08:44:56.020922Z","shell.execute_reply.started":"2022-08-12T08:44:55.971846Z","shell.execute_reply":"2022-08-12T08:44:56.019551Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = train_df.drop(columns=['SalePrice', 'Id'])\ny = train_df['SalePrice']\n\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.15, random_state=1994)","metadata":{"execution":{"iopub.status.busy":"2022-08-12T08:44:56.768266Z","iopub.execute_input":"2022-08-12T08:44:56.768932Z","iopub.status.idle":"2022-08-12T08:44:56.780296Z","shell.execute_reply.started":"2022-08-12T08:44:56.768887Z","shell.execute_reply":"2022-08-12T08:44:56.779079Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"I concat both datasets to save time and apply the same preprocessing, I won't feed any test data to our algos though until the time of final submission","metadata":{}},{"cell_type":"code","source":"data = pd.concat([train_df, test_df], axis=0, ignore_index=True)\n\nX = data.drop(columns=['SalePrice', 'Id'])\ny = data['SalePrice']\n\nX_train_data = X[:1460]\nX_test_data = X[1460:]\ny_train_data = y[:1460]\ny_test_data = y[1460:]\n\nX_train, X_test, y_train, y_test = train_test_split(X_train_data, y_train_data, test_size=0.15, random_state=1994)","metadata":{"execution":{"iopub.status.busy":"2022-08-12T08:44:59.438071Z","iopub.execute_input":"2022-08-12T08:44:59.438485Z","iopub.status.idle":"2022-08-12T08:44:59.471330Z","shell.execute_reply.started":"2022-08-12T08:44:59.438454Z","shell.execute_reply":"2022-08-12T08:44:59.470161Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data[:1460].tail() # train data","metadata":{"execution":{"iopub.status.busy":"2022-08-12T08:45:00.526464Z","iopub.execute_input":"2022-08-12T08:45:00.527724Z","iopub.status.idle":"2022-08-12T08:45:00.556220Z","shell.execute_reply.started":"2022-08-12T08:45:00.527674Z","shell.execute_reply":"2022-08-12T08:45:00.555310Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data[1460:].head() # test_data","metadata":{"execution":{"iopub.status.busy":"2022-08-12T08:45:00.821019Z","iopub.execute_input":"2022-08-12T08:45:00.821451Z","iopub.status.idle":"2022-08-12T08:45:00.849821Z","shell.execute_reply.started":"2022-08-12T08:45:00.821417Z","shell.execute_reply":"2022-08-12T08:45:00.848424Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Quick EDA","metadata":{}},{"cell_type":"code","source":"# Correlation Matrix\n\nf, ax = plt.subplots(figsize=(30, 25))\nmat = train_df.corr('pearson')\nmask = np.triu(np.ones_like(mat, dtype=bool))\ncmap = sns.diverging_palette(230, 20, as_cmap=True)\nsns.heatmap(mat, mask=mask, cmap=cmap, vmax=1, center=0, annot = True,\n            square=True, linewidths=.5, cbar_kws={\"shrink\": .5})\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-12T08:45:02.001437Z","iopub.execute_input":"2022-08-12T08:45:02.002320Z","iopub.status.idle":"2022-08-12T08:45:05.660015Z","shell.execute_reply.started":"2022-08-12T08:45:02.002260Z","shell.execute_reply":"2022-08-12T08:45:05.658860Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* Fireplaces\n* MasVnrArea\n* GarageYrBlt\n* YearRemodAdd\n* YearBuilt\n* TotRmsAbvGrd\n* FullBath\n* 1stFlrSF\n* TotalBsmtSF    \n* GarageArea   \n* GarageCars \n* GrLivArea \n* OverallQual \n* SalePrice  ","metadata":{}},{"cell_type":"markdown","source":"Just to be sure I am not dropping any columns that are strong positive or negative correlated to the target variable","metadata":{}},{"cell_type":"code","source":"mat['SalePrice'].sort_values()","metadata":{"execution":{"iopub.status.busy":"2022-08-12T08:09:04.978127Z","iopub.execute_input":"2022-08-12T08:09:04.979326Z","iopub.status.idle":"2022-08-12T08:09:04.992309Z","shell.execute_reply.started":"2022-08-12T08:09:04.979278Z","shell.execute_reply":"2022-08-12T08:09:04.991350Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Getting the main parameters of the Normal Ditribution ()\n(mu, sigma) = norm.fit(train_df['SalePrice'])\n\nplt.figure(figsize = (12,6))\nsns.distplot(train_df['SalePrice'], kde = True, hist=True, fit = norm)\nplt.title('SalePrice distribution vs Normal Distribution', fontsize = 13)\nplt.xlabel(\"House's sale Price in $\", fontsize = 12)\nplt.legend(['Normal dist. ($\\mu=$ {:.2f} and $\\sigma=$ {:.2f} )'.format(mu, sigma)],\n            loc='best')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-12T08:09:04.993653Z","iopub.execute_input":"2022-08-12T08:09:04.994249Z","iopub.status.idle":"2022-08-12T08:09:05.552294Z","shell.execute_reply.started":"2022-08-12T08:09:04.994206Z","shell.execute_reply":"2022-08-12T08:09:05.551022Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f\"The skewness of the target variable in train set is: {skew(y_train)}\")\nprint(f\"The kurtosis of the target variable in train set is: {kurtosis(y_train)}\")","metadata":{"execution":{"iopub.status.busy":"2022-08-12T08:09:05.554752Z","iopub.execute_input":"2022-08-12T08:09:05.555158Z","iopub.status.idle":"2022-08-12T08:09:05.562896Z","shell.execute_reply.started":"2022-08-12T08:09:05.555097Z","shell.execute_reply":"2022-08-12T08:09:05.561646Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**TODO**: log transformation on target variable, highly skewed","metadata":{}},{"cell_type":"markdown","source":"#### Missing values","metadata":{}},{"cell_type":"code","source":"missing_ratio = X.isna().mean().round(4) * 100\nmissing_ratio.sort_values(ascending=False)[:15]","metadata":{"execution":{"iopub.status.busy":"2022-08-12T08:33:28.665773Z","iopub.execute_input":"2022-08-12T08:33:28.667093Z","iopub.status.idle":"2022-08-12T08:33:28.685751Z","shell.execute_reply.started":"2022-08-12T08:33:28.667001Z","shell.execute_reply":"2022-08-12T08:33:28.684412Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Filling Alley, PoolQC, Fence and MiscFeature accordingly - those NAs are not really NA. The data dictionary maps those NAs with values such as \"No alley\" ,\"No pool\", etc. Therefore that's how I will impute them.**","metadata":{}},{"cell_type":"code","source":"# Fill missing values \n\nvalues = {\"Alley\": 'No Alley Access', \"PoolQC\": \"No Pool\", \"Fence\": \"No Fence\", \"MiscFeature\": \"None\"}\n\nX_train.fillna(value=values, inplace=True)\nX_test.fillna(value=values, inplace=True)\n\nX_test_data.fillna(value=values, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-12T08:45:15.739803Z","iopub.execute_input":"2022-08-12T08:45:15.740231Z","iopub.status.idle":"2022-08-12T08:45:15.751990Z","shell.execute_reply.started":"2022-08-12T08:45:15.740195Z","shell.execute_reply":"2022-08-12T08:45:15.751130Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**LotFrontage, MasVnrArea and GarageYrBlt are numerical values with very little missing values, therefore we can safely impute those with the mean**","metadata":{}},{"cell_type":"code","source":"imputer = SimpleImputer(strategy='mean')\n\nna_cols = ['LotFrontage', 'MasVnrArea', 'GarageYrBlt']\n\ntrain_imputed_values = imputer.fit_transform(X_train[na_cols])\ntest_imputed_values = imputer.transform(X_test[na_cols])\n\nX_train[na_cols] = train_imputed_values\nX_test[na_cols] = test_imputed_values\n\n# Test data\nX_test_data_imputed_values = imputer.transform(X_test_data[na_cols])\nX_test_data[na_cols] = X_test_data_imputed_values","metadata":{"execution":{"iopub.status.busy":"2022-08-12T08:45:16.772397Z","iopub.execute_input":"2022-08-12T08:45:16.772827Z","iopub.status.idle":"2022-08-12T08:45:16.792941Z","shell.execute_reply.started":"2022-08-12T08:45:16.772791Z","shell.execute_reply":"2022-08-12T08:45:16.791699Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train['GarageYrBlt'] = round(X_train['GarageYrBlt'])\nX_test['GarageYrBlt'] = round(X_test['GarageYrBlt'])\n\nX_train['Exterior2nd'] = X_train['Exterior2nd'].replace({'Brk Cmn': 'BrkComm'})\nX_test['Exterior2nd'] = X_test['Exterior2nd'].replace({'Brk Cmn': 'BrkComm'})\n\n# Test data\nX_test_data['GarageYrBlt'] = round(X_test_data['GarageYrBlt'])\nX_test_data['Exterior2nd'] = X_test_data['Exterior2nd'].replace({'Brk Cmn': 'BrkComm'})","metadata":{"execution":{"iopub.status.busy":"2022-08-12T08:45:19.819869Z","iopub.execute_input":"2022-08-12T08:45:19.820643Z","iopub.status.idle":"2022-08-12T08:45:19.835017Z","shell.execute_reply.started":"2022-08-12T08:45:19.820584Z","shell.execute_reply":"2022-08-12T08:45:19.833661Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Still have a few missing values, but those columns don't even look that important to deal with, therefore i'm dropping them and maybe deal with them later if I have to ","metadata":{}},{"cell_type":"code","source":"drop_cols = [col for col in X_train.columns if X_train[col].isnull().any()]","metadata":{"_kg_hide-input":false,"execution":{"iopub.status.busy":"2022-08-12T08:45:23.567755Z","iopub.execute_input":"2022-08-12T08:45:23.568173Z","iopub.status.idle":"2022-08-12T08:45:23.594289Z","shell.execute_reply.started":"2022-08-12T08:45:23.568128Z","shell.execute_reply":"2022-08-12T08:45:23.593223Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train = X_train.drop(columns=drop_cols)\nX_test = X_test.drop(columns=drop_cols)","metadata":{"execution":{"iopub.status.busy":"2022-08-12T08:45:29.057659Z","iopub.execute_input":"2022-08-12T08:45:29.058070Z","iopub.status.idle":"2022-08-12T08:45:29.067271Z","shell.execute_reply.started":"2022-08-12T08:45:29.058010Z","shell.execute_reply":"2022-08-12T08:45:29.065613Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# test_Data \nX_test_data = X_test_data.drop(columns=drop_cols)","metadata":{"execution":{"iopub.status.busy":"2022-08-12T08:45:31.586123Z","iopub.execute_input":"2022-08-12T08:45:31.586562Z","iopub.status.idle":"2022-08-12T08:45:31.593825Z","shell.execute_reply.started":"2022-08-12T08:45:31.586526Z","shell.execute_reply":"2022-08-12T08:45:31.592900Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Fill remaining missing values","metadata":{}},{"cell_type":"code","source":"X_test_data['MSZoning'] = X_test_data['MSZoning'].fillna(value='RL')\nX_test_data['Utilities'] = X_test_data['Utilities'].fillna(value='AllPub')\nX_test_data['Exterior1st'] = X_test_data['Exterior1st'].fillna(value='VinylSd')\nX_test_data['Exterior2nd'] = X_test_data['Exterior2nd'].fillna(value='VinylSd')\n\nX_test_data['BsmtFinSF1'] = X_test_data['BsmtFinSF1'].fillna(value=0)\nX_test_data['BsmtFinSF2'] = X_test_data['BsmtFinSF2'].fillna(value=0)\nX_test_data['BsmtUnfSF'] = X_test_data['BsmtUnfSF'].fillna(value=0)\nX_test_data['TotalBsmtSF'] = X_test_data['TotalBsmtSF'].fillna(value=0)\nX_test_data['BsmtFullBath'] = X_test_data['BsmtFullBath'].fillna(value=0)\nX_test_data['BsmtHalfBath'] = X_test_data['BsmtHalfBath'].fillna(value=0)\n\nX_test_data['KitchenQual'] = X_test_data['KitchenQual'].fillna(value='TA')\nX_test_data['Functional'] = X_test_data['Functional'].fillna(value='Typ')\n\nX_test_data['GarageCars'] = X_test_data['GarageCars'].fillna(value=0)\nX_test_data['GarageArea'] = X_test_data['GarageArea'].fillna(value=0)\n\nX_test_data['SaleType'] = X_test_data['SaleType'].fillna(value='WD')","metadata":{"execution":{"iopub.status.busy":"2022-08-12T08:50:51.048543Z","iopub.execute_input":"2022-08-12T08:50:51.048976Z","iopub.status.idle":"2022-08-12T08:50:51.068477Z","shell.execute_reply.started":"2022-08-12T08:50:51.048942Z","shell.execute_reply":"2022-08-12T08:50:51.067421Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Train columns with missing data:')\n\nfor k,v in X_train.isnull().sum().to_dict().items():\n  if v != 0:\n    print(f\"{k}:{v}\")\n  else:\n    continue\n\nprint('---------------------------------')\nprint('Test columns with missing data')\n\nfor k,v in X_test.isnull().sum().to_dict().items():\n  if v != 0:\n    print(f\"{k}:{v}\")\n  else:\n    continue","metadata":{"execution":{"iopub.status.busy":"2022-08-12T08:45:44.538001Z","iopub.execute_input":"2022-08-12T08:45:44.538416Z","iopub.status.idle":"2022-08-12T08:45:44.555992Z","shell.execute_reply.started":"2022-08-12T08:45:44.538383Z","shell.execute_reply":"2022-08-12T08:45:44.554908Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Sanity check on dimensions, making sure I did not ditch any columns I shouldn't**","metadata":{}},{"cell_type":"code","source":"print(X_train.shape)\nprint(X_test.shape)\nprint(X_test_data.shape)","metadata":{"execution":{"iopub.status.busy":"2022-08-12T08:52:03.978256Z","iopub.execute_input":"2022-08-12T08:52:03.978811Z","iopub.status.idle":"2022-08-12T08:52:03.985288Z","shell.execute_reply.started":"2022-08-12T08:52:03.978764Z","shell.execute_reply":"2022-08-12T08:52:03.984238Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Dealing with categorical columns","metadata":{}},{"cell_type":"markdown","source":"#### Ordinal Encoding","metadata":{}},{"cell_type":"code","source":"ord_encoder = OrdinalEncoder(categories=[['Po', 'Fa', 'TA', 'Gd', 'Ex']])\n\nX_train['ExterQual'] = ord_encoder.fit_transform(X_train[['ExterQual']])\nX_test['ExterQual'] = ord_encoder.transform(X_test[['ExterQual']])\n\nX_train['ExterCond'] = ord_encoder.fit_transform(X_train[['ExterCond']])\nX_test['ExterCond'] = ord_encoder.transform(X_test[['ExterCond']])\n\nX_train['HeatingQC'] = ord_encoder.fit_transform(X_train[['HeatingQC']])\nX_test['HeatingQC'] = ord_encoder.transform(X_test[['HeatingQC']])\n\nX_train['KitchenQual'] = ord_encoder.fit_transform(X_train[['KitchenQual']])\nX_test['KitchenQual'] = ord_encoder.transform(X_test[['KitchenQual']])","metadata":{"execution":{"iopub.status.busy":"2022-08-12T08:53:19.876232Z","iopub.execute_input":"2022-08-12T08:53:19.876712Z","iopub.status.idle":"2022-08-12T08:53:19.902521Z","shell.execute_reply.started":"2022-08-12T08:53:19.876676Z","shell.execute_reply":"2022-08-12T08:53:19.901544Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_test_data['ExterQual'] = ord_encoder.transform(X_test_data[['ExterQual']])\nX_test_data['ExterCond'] = ord_encoder.transform(X_test_data[['ExterCond']])\nX_test_data['HeatingQC'] = ord_encoder.transform(X_test_data[['HeatingQC']])\nX_test_data['KitchenQual'] = ord_encoder.transform(X_test_data[['KitchenQual']])","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Target Encoding","metadata":{}},{"cell_type":"code","source":"categorical_cols = [col for col in X_train.columns if X_train[col].dtypes == 'object']\n\ncategorical_cols = [col for col in categorical_cols if col not in ['KitchenQual', 'HeatingQC', 'ExterCond', 'ExterQual']]\n\ntarget_encoder = MEstimateEncoder(cols = X_train[categorical_cols], m=5.0)\n\nX_train = target_encoder.fit_transform(X_train, y_train)\n\nX_test = target_encoder.transform(X_test)","metadata":{"execution":{"iopub.status.busy":"2022-08-12T08:58:27.080249Z","iopub.execute_input":"2022-08-12T08:58:27.080642Z","iopub.status.idle":"2022-08-12T08:58:27.656405Z","shell.execute_reply.started":"2022-08-12T08:58:27.080611Z","shell.execute_reply":"2022-08-12T08:58:27.655281Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_test_data = target_encoder.transform(X_test_data)","metadata":{"execution":{"iopub.status.busy":"2022-08-12T08:58:30.927866Z","iopub.execute_input":"2022-08-12T08:58:30.928280Z","iopub.status.idle":"2022-08-12T08:58:30.999059Z","shell.execute_reply.started":"2022-08-12T08:58:30.928249Z","shell.execute_reply":"2022-08-12T08:58:30.997816Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Feature Engineering","metadata":{}},{"cell_type":"code","source":"X_train['Total_Living'] = X_train['GrLivArea'] + X_train['TotalBsmtSF'] + X_train['GarageArea']\n\nX_test['Total_Living'] = X_test['GrLivArea'] + X_test['TotalBsmtSF'] + X_test['GarageArea']","metadata":{"execution":{"iopub.status.busy":"2022-08-12T08:58:54.888348Z","iopub.execute_input":"2022-08-12T08:58:54.888742Z","iopub.status.idle":"2022-08-12T08:58:54.897428Z","shell.execute_reply.started":"2022-08-12T08:58:54.888711Z","shell.execute_reply":"2022-08-12T08:58:54.896126Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_test_data['Total_Living'] = X_test_data['GrLivArea'] + X_test_data['TotalBsmtSF'] + X_test_data['GarageArea']","metadata":{"execution":{"iopub.status.busy":"2022-08-12T08:58:58.538206Z","iopub.execute_input":"2022-08-12T08:58:58.538607Z","iopub.status.idle":"2022-08-12T08:58:58.546416Z","shell.execute_reply.started":"2022-08-12T08:58:58.538574Z","shell.execute_reply":"2022-08-12T08:58:58.545072Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Standard Scaling for numerical variables ","metadata":{}},{"cell_type":"code","source":"numerical_cols = [col for col in X_train.columns if X_train[col].dtypes != 'object']\n\nscaler = StandardScaler()\nX_train_scaled = scaler.fit_transform(X_train[numerical_cols])\nX_test_scaled = scaler.transform(X_test[numerical_cols])\n\nX_train[numerical_cols] = X_train_scaled\nX_test[numerical_cols] = X_test_scaled","metadata":{"execution":{"iopub.status.busy":"2022-08-12T08:59:26.557912Z","iopub.execute_input":"2022-08-12T08:59:26.559063Z","iopub.status.idle":"2022-08-12T08:59:26.594536Z","shell.execute_reply.started":"2022-08-12T08:59:26.558992Z","shell.execute_reply":"2022-08-12T08:59:26.593672Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_test_data_scaled = scaler.transform(X_test_data[numerical_cols])\nX_test_data[numerical_cols] = X_test_data_scaled","metadata":{"execution":{"iopub.status.busy":"2022-08-12T08:59:30.727992Z","iopub.execute_input":"2022-08-12T08:59:30.728381Z","iopub.status.idle":"2022-08-12T08:59:30.747357Z","shell.execute_reply.started":"2022-08-12T08:59:30.728351Z","shell.execute_reply":"2022-08-12T08:59:30.746098Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Scaling target variable\n\nRemember that target variable was highly skewed!!","metadata":{}},{"cell_type":"code","source":"y_train_log = np.log10(y_train)\ny_test_log = np.log10(y_test)","metadata":{"execution":{"iopub.status.busy":"2022-08-12T08:59:38.886280Z","iopub.execute_input":"2022-08-12T08:59:38.886821Z","iopub.status.idle":"2022-08-12T08:59:38.893226Z","shell.execute_reply.started":"2022-08-12T08:59:38.886776Z","shell.execute_reply":"2022-08-12T08:59:38.891744Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Modelling","metadata":{}},{"cell_type":"markdown","source":"To avoid overfitting, I'm always using some kind of ensembling. For the final regressor I will be using stackingregressor from sklearn and as my estimators I'm using ridge, support vector regression, xgboost, huber regressor and CatBoostRegressor","metadata":{}},{"cell_type":"markdown","source":"**Estimators**","metadata":{}},{"cell_type":"code","source":"# ridge \nridgemodel = Ridge(alpha=26)\n\n# xgbmodel\nxgbmodel = XGBRegressor(objective='reg:squarederror', \n                        n_estimators=1000, \n                        learning_rate=0.02)\n\n# SVR \nsvrmodel = SVR(C=8, \n               epsilon=0.00005, \n               gamma=0.0008)\n\n# Huber \nhubermodel = HuberRegressor(alpha=30,\n                            epsilon=3,\n                            fit_intercept=True,\n                            max_iter=2000)\n\n# CatBoost\ncbmodel = cb.CatBoostRegressor(loss_function='RMSE',\n                               colsample_bylevel=0.3, \n                               depth=2,\n                               l2_leaf_reg=20, \n                               learning_rate=0.005, \n                               n_estimators=15000, \n                               subsample=0.3,\n                               verbose=False)","metadata":{"execution":{"iopub.status.busy":"2022-08-12T08:59:43.038410Z","iopub.execute_input":"2022-08-12T08:59:43.038811Z","iopub.status.idle":"2022-08-12T08:59:43.047175Z","shell.execute_reply.started":"2022-08-12T08:59:43.038771Z","shell.execute_reply":"2022-08-12T08:59:43.045518Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"estimators = [('ridgemodel', Ridge(alpha=26)), \n              ('svrmodel', SVR(C=8, epsilon=0.00005, gamma=0.0008)), \n              ('hubermodel', HuberRegressor(alpha=30,epsilon=3,fit_intercept=True,max_iter=10000)), \n              ('cbmodel', cb.CatBoostRegressor(loss_function='RMSE',colsample_bylevel=0.3, depth=2,\n                          l2_leaf_reg=20, learning_rate=0.005, n_estimators=15000, subsample=0.3,verbose=False))]\n\nstackmodel = StackingRegressor(estimators=estimators, final_estimator=XGBRegressor(objective='reg:squarederror', n_estimators=1000, learning_rate=0.02))","metadata":{"execution":{"iopub.status.busy":"2022-08-12T08:59:48.826576Z","iopub.execute_input":"2022-08-12T08:59:48.827059Z","iopub.status.idle":"2022-08-12T08:59:48.835580Z","shell.execute_reply.started":"2022-08-12T08:59:48.826998Z","shell.execute_reply":"2022-08-12T08:59:48.834717Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"-----------------------------\")\nprint(\"Overview of model performance\")\nprint(\"-----------------------------\")\nfor i in [ridgemodel,hubermodel,cbmodel,svrmodel,xgbmodel,stackmodel]:\n    i.fit(X_train, y_train_log)\n\n    print(f\"For model {i}\")\n    print(f\"Train RMSE: {round(math.sqrt(mse(y_train_log,i.predict(X_train))), 4)}\")\n    print(f\"Test RMSE: {round(math.sqrt(mse(y_test_log,i.predict(X_test))), 4)}\")\n    print(\"-----------------------------\")\nprint(\"-----------------------------\")\nprint(\"Average Test RMSE for ensemble model: \")\nfit = (svrmodel.predict(X_test) + xgbmodel.predict(X_test) + stackmodel.predict(X_test) + ridgemodel.predict(X_test) + hubermodel.predict(X_test) + cbmodel.predict(X_test)) / 6\nprint(math.sqrt(mse(y_test_log,fit)))","metadata":{"execution":{"iopub.status.busy":"2022-08-12T08:59:52.907899Z","iopub.execute_input":"2022-08-12T08:59:52.908324Z","iopub.status.idle":"2022-08-12T09:00:59.659732Z","shell.execute_reply.started":"2022-08-12T08:59:52.908288Z","shell.execute_reply":"2022-08-12T09:00:59.658096Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"final_prediction = (ridgemodel.predict(X_test) + 3 * xgbmodel.predict(X_test) \\\n+  5 * stackmodel.predict(X_test) + 4 * svrmodel.predict(X_test) \\\n+  hubermodel.predict(X_test) +  cbmodel.predict(X_test)) / 15\n\nround(math.sqrt(mse(y_test_log,final_prediction)), 4)","metadata":{"execution":{"iopub.status.busy":"2022-08-12T09:01:09.210772Z","iopub.execute_input":"2022-08-12T09:01:09.211479Z","iopub.status.idle":"2022-08-12T09:01:09.525785Z","shell.execute_reply.started":"2022-08-12T09:01:09.211441Z","shell.execute_reply":"2022-08-12T09:01:09.524124Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Submission**","metadata":{}},{"cell_type":"code","source":"# For now, this is the most stable mix of the predictors - nothing fancy, just trial and error\ntest_prediction = (ridgemodel.predict(X_test_data) + 3 * xgbmodel.predict(X_test_data) \\\n+  5 * stackmodel.predict(X_test_data) + 4 * svrmodel.predict(X_test_data) \\\n+  hubermodel.predict(X_test_data) +  cbmodel.predict(X_test_data)) / 15","metadata":{"execution":{"iopub.status.busy":"2022-08-12T10:08:23.366974Z","iopub.execute_input":"2022-08-12T10:08:23.367502Z","iopub.status.idle":"2022-08-12T10:08:24.105345Z","shell.execute_reply.started":"2022-08-12T10:08:23.367451Z","shell.execute_reply":"2022-08-12T10:08:24.103518Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# reverse the log10 transformation\nsample_submission['SalePrice'] = np.array(10**test_prediction) ","metadata":{"execution":{"iopub.status.busy":"2022-08-12T09:04:19.627266Z","iopub.execute_input":"2022-08-12T09:04:19.627665Z","iopub.status.idle":"2022-08-12T09:04:19.633322Z","shell.execute_reply.started":"2022-08-12T09:04:19.627632Z","shell.execute_reply":"2022-08-12T09:04:19.632120Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Always fingers crossed before submission \nsample_submission.to_csv(\"submission.csv\", index = False, header = True)","metadata":{"execution":{"iopub.status.busy":"2022-08-12T09:05:19.148109Z","iopub.execute_input":"2022-08-12T09:05:19.148477Z","iopub.status.idle":"2022-08-12T09:05:19.161935Z","shell.execute_reply.started":"2022-08-12T09:05:19.148448Z","shell.execute_reply":"2022-08-12T09:05:19.160550Z"},"trusted":true},"execution_count":null,"outputs":[]}]}