{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Content\n\n# 1 - Reading the data\n\n# 2 - Preprocessing\n\n+ **2.1. Dealing with Missing Values**\n+ **2.2. Target Encoding**\n+ **2.3.  Feature Engineering**\n+ **2.4.  Standard Scaling**\n     \n# 3 - Modeling with Stacking\n\n# 4 - Making a Submission","metadata":{}},{"cell_type":"markdown","source":"**Reading the data from Kaggle**","metadata":{}},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-08T15:56:07.679783Z","iopub.execute_input":"2022-07-08T15:56:07.680244Z","iopub.status.idle":"2022-07-08T15:56:07.710712Z","shell.execute_reply.started":"2022-07-08T15:56:07.680155Z","shell.execute_reply":"2022-07-08T15:56:07.70981Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Deactivating Errors**","metadata":{}},{"cell_type":"code","source":"pd.options.mode.chained_assignment = None  # default='warn'","metadata":{"execution":{"iopub.status.busy":"2022-07-08T15:56:07.712329Z","iopub.execute_input":"2022-07-08T15:56:07.712669Z","iopub.status.idle":"2022-07-08T15:56:07.719814Z","shell.execute_reply.started":"2022-07-08T15:56:07.71264Z","shell.execute_reply":"2022-07-08T15:56:07.718597Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Reading the data**","metadata":{}},{"cell_type":"code","source":"train_df = pd.read_csv(\"/kaggle/input/house-prices-advanced-regression-techniques/train.csv\")\ntest_df = pd.read_csv(\"/kaggle/input/house-prices-advanced-regression-techniques/test.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-07-08T15:56:07.722213Z","iopub.execute_input":"2022-07-08T15:56:07.722607Z","iopub.status.idle":"2022-07-08T15:56:07.788546Z","shell.execute_reply.started":"2022-07-08T15:56:07.722576Z","shell.execute_reply":"2022-07-08T15:56:07.787521Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Importing the Modules**","metadata":{}},{"cell_type":"code","source":"!pip install category_encoders\n!pip install catboost\n\nimport pandas as pd\nimport numpy as np\nimport io\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.feature_selection import mutual_info_regression\nfrom sklearn.ensemble import GradientBoostingRegressor\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.model_selection import GridSearchCV\nfrom sklearn.preprocessing import OrdinalEncoder\nfrom sklearn.preprocessing import OneHotEncoder\nfrom category_encoders import MEstimateEncoder, TargetEncoder\nfrom xgboost import XGBRegressor\nfrom sklearn.ensemble import RandomForestRegressor, StackingRegressor, ExtraTreesRegressor\nfrom sklearn.linear_model import Ridge, HuberRegressor, LinearRegression\nfrom sklearn.svm import SVR\nimport catboost as cb\nfrom sklearn.metrics import mean_absolute_error as mae\nfrom sklearn.metrics import mean_squared_error as mse\nfrom sklearn.model_selection import cross_val_score\nfrom sklearn import preprocessing\nfrom sklearn.metrics import r2_score\nfrom sklearn.model_selection import KFold\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nimport math\n\n","metadata":{"execution":{"iopub.status.busy":"2022-07-08T15:57:11.850214Z","iopub.execute_input":"2022-07-08T15:57:11.8506Z","iopub.status.idle":"2022-07-08T15:57:33.802029Z","shell.execute_reply.started":"2022-07-08T15:57:11.850568Z","shell.execute_reply":"2022-07-08T15:57:33.800982Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Creating a run_model function to use later\n\ndef run_model(model, X_train, X_test, y_train, y_test):\n  model.fit(X_train, y_train)\n\n  preds = model.predict(X_test)\n\n  print(f\"Test RMSE: {round(math.sqrt(mse(y_test, preds)),4)}\")\n  print(f\"Test R2: {round(r2_score(y_test, preds),4)}\")\n\n  print('-------------------------------------------------')\n\n  preds_train = model.predict(X_train)\n\n  print(f\"Train RMSE: {round(math.sqrt(mse(y_train, preds_train)),4)}\")\n  print(f\"Train R2: {round(r2_score(y_train, preds_train),4)}\")","metadata":{"execution":{"iopub.status.busy":"2022-07-08T16:10:22.247996Z","iopub.execute_input":"2022-07-08T16:10:22.248446Z","iopub.status.idle":"2022-07-08T16:10:22.255596Z","shell.execute_reply.started":"2022-07-08T16:10:22.248396Z","shell.execute_reply":"2022-07-08T16:10:22.254318Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Splitting the Dataset**","metadata":{}},{"cell_type":"code","source":"X = train_df.drop(columns=['SalePrice', 'Id'])\ny = train_df['SalePrice']\n\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.15, random_state=38)","metadata":{"execution":{"iopub.status.busy":"2022-07-08T15:57:36.89801Z","iopub.execute_input":"2022-07-08T15:57:36.899014Z","iopub.status.idle":"2022-07-08T15:57:36.921391Z","shell.execute_reply.started":"2022-07-08T15:57:36.898978Z","shell.execute_reply":"2022-07-08T15:57:36.920483Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**NOTE:** To avoid data leakage, all the preprocessing steps will be performed on training and test data separately. Hence, the first thing we do is split the data.","metadata":{}},{"cell_type":"markdown","source":"# **Pre-Processing Steps (Imputing, Encoding, Standardizing)**","metadata":{}},{"cell_type":"markdown","source":"**Dealing with Missing Values**","metadata":{}},{"cell_type":"markdown","source":"First we look at the numerical and categorical columns which have missing values.","metadata":{}},{"cell_type":"code","source":"numerical_cols = [col for col in X.columns if X[col].dtypes != 'object']\ncategorical_cols = [col for col in X.columns if X[col].dtypes == 'object']\n\nnumerical_na_cols = []\ncategorical_na_cols = []\n\nfor k,v in X[numerical_cols].isnull().sum().to_dict().items():\n  if v != 0:\n    numerical_na_cols.append(k)\n\nfor k,v in X[categorical_cols].isnull().sum().to_dict().items():\n  if v != 0:\n    categorical_na_cols.append(k)\n\nprint(numerical_na_cols)\nprint(categorical_na_cols)","metadata":{"execution":{"iopub.status.busy":"2022-07-08T15:57:51.937284Z","iopub.execute_input":"2022-07-08T15:57:51.937745Z","iopub.status.idle":"2022-07-08T15:57:51.957863Z","shell.execute_reply.started":"2022-07-08T15:57:51.93771Z","shell.execute_reply":"2022-07-08T15:57:51.956661Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"For numerical columns with missing values, I use SimpleImputer to replace the missing values with mean. To avoid data leakage, imputing is applied to train and test data separately.","metadata":{}},{"cell_type":"code","source":"# Imputing numerical columns with missing data \n\nmy_imputer = SimpleImputer(strategy='mean')\n\ntrain_imputed_values = my_imputer.fit_transform(X_train[numerical_na_cols])\ntest_imputed_values = my_imputer.transform(X_test[numerical_na_cols])\n\nX_train[numerical_na_cols] = train_imputed_values\nX_test[numerical_na_cols] = test_imputed_values","metadata":{"execution":{"iopub.status.busy":"2022-07-08T15:58:07.499464Z","iopub.execute_input":"2022-07-08T15:58:07.499884Z","iopub.status.idle":"2022-07-08T15:58:07.517628Z","shell.execute_reply.started":"2022-07-08T15:58:07.499849Z","shell.execute_reply":"2022-07-08T15:58:07.516394Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Just cleaning some data in GarageYrBlt and Exterior2nd columns.","metadata":{}},{"cell_type":"code","source":"X_train['GarageYrBlt'] = round(X_train['GarageYrBlt'])\nX_test['GarageYrBlt'] = round(X_test['GarageYrBlt'])\n\nX_train['Exterior2nd'] = X_train['Exterior2nd'].replace({'Brk Cmn': 'BrkComm'})\nX_test['Exterior2nd'] = X_test['Exterior2nd'].replace({'Brk Cmn': 'BrkComm'})\ntest_df['Exterior2nd'] = test_df['Exterior2nd'].replace({'Brk Cmn': 'BrkComm'})","metadata":{"execution":{"iopub.status.busy":"2022-07-08T15:59:11.014964Z","iopub.execute_input":"2022-07-08T15:59:11.015323Z","iopub.status.idle":"2022-07-08T15:59:11.026127Z","shell.execute_reply.started":"2022-07-08T15:59:11.015294Z","shell.execute_reply":"2022-07-08T15:59:11.024949Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"For Alley, PoolQC, Fense and MiscFeature columns, I simply replace the null values with the description given in data_description.txt file. This file is accessible in Data tab of the competition. ","metadata":{}},{"cell_type":"code","source":"values = {\"Alley\": 'No Alley Access', \"PoolQC\": \"No Pool\", \"Fence\": \"No Fence\", \"MiscFeature\": \"None\"}\n\nX_train.fillna(value=values, inplace=True)\nX_test.fillna(value=values, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-08T15:59:17.569436Z","iopub.execute_input":"2022-07-08T15:59:17.569809Z","iopub.status.idle":"2022-07-08T15:59:17.579703Z","shell.execute_reply.started":"2022-07-08T15:59:17.569778Z","shell.execute_reply":"2022-07-08T15:59:17.578262Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"For now, I choose to drop the rest of the columns since it will take a bit of time to individually analyze each one of them to select the correct filling method. After running the model, we might come back to this step if we still want to decrease the RMSE.","metadata":{}},{"cell_type":"code","source":"drop_cols = [col for col in X_train.columns if X_train[col].isnull().any()]\n\nX_train = X_train.drop(columns=drop_cols)\nX_test = X_test.drop(columns=drop_cols)","metadata":{"execution":{"iopub.status.busy":"2022-07-08T16:01:36.124393Z","iopub.execute_input":"2022-07-08T16:01:36.124838Z","iopub.status.idle":"2022-07-08T16:01:36.154314Z","shell.execute_reply.started":"2022-07-08T16:01:36.124807Z","shell.execute_reply":"2022-07-08T16:01:36.152961Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Checking if there is still any columns with missing values in both train and test data.","metadata":{}},{"cell_type":"code","source":"# Checking columns with missing values\n\nprint('Train columns with missing data:')\n\nfor k,v in X_train.isnull().sum().to_dict().items():\n  if v != 0:\n    print(f\"{k}:{v}\")\n  else:\n    continue\n\nprint('---------------------------------')\nprint('Test columns with missing data')\n\nfor k,v in X_test.isnull().sum().to_dict().items():\n  if v != 0:\n    print(f\"{k}:{v}\")\n  else:\n    continue","metadata":{"execution":{"iopub.status.busy":"2022-07-08T16:01:43.023576Z","iopub.execute_input":"2022-07-08T16:01:43.024134Z","iopub.status.idle":"2022-07-08T16:01:43.046541Z","shell.execute_reply.started":"2022-07-08T16:01:43.024086Z","shell.execute_reply":"2022-07-08T16:01:43.045091Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"There is no missing value left. Now we can continue with encoding. First we log-transform the target variable. After that, we will encode the categorical variables using both one-hot encoding and target encoding, and compare the results.","metadata":{}},{"cell_type":"markdown","source":"**Log-Transformation of Target Variable**","metadata":{}},{"cell_type":"code","source":"y_train_log = np.log10(y_train)\ny_test_log = np.log10(y_test)","metadata":{"execution":{"iopub.status.busy":"2022-07-08T16:02:05.497507Z","iopub.execute_input":"2022-07-08T16:02:05.497899Z","iopub.status.idle":"2022-07-08T16:02:05.503291Z","shell.execute_reply.started":"2022-07-08T16:02:05.497871Z","shell.execute_reply":"2022-07-08T16:02:05.502455Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Ordinal Encoding**","metadata":{}},{"cell_type":"code","source":"ord_encoder = OrdinalEncoder(categories=[['Po', 'Fa', 'TA', 'Gd', 'Ex']])\n\nX_train['ExterQual'] = ord_encoder.fit_transform(X_train[['ExterQual']])\nX_test['ExterQual'] = ord_encoder.transform(X_test[['ExterQual']])\n\nX_train['ExterCond'] = ord_encoder.fit_transform(X_train[['ExterCond']])\nX_test['ExterCond'] = ord_encoder.transform(X_test[['ExterCond']])\n\nX_train['HeatingQC'] = ord_encoder.fit_transform(X_train[['HeatingQC']])\nX_test['HeatingQC'] = ord_encoder.transform(X_test[['HeatingQC']])\n\nX_train['KitchenQual'] = ord_encoder.fit_transform(X_train[['KitchenQual']])\nX_test['KitchenQual'] = ord_encoder.transform(X_test[['KitchenQual']])","metadata":{"execution":{"iopub.status.busy":"2022-07-08T16:02:32.539958Z","iopub.execute_input":"2022-07-08T16:02:32.540327Z","iopub.status.idle":"2022-07-08T16:02:32.565363Z","shell.execute_reply.started":"2022-07-08T16:02:32.540297Z","shell.execute_reply":"2022-07-08T16:02:32.563713Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Target Encoding**","metadata":{}},{"cell_type":"code","source":"categorical_cols = [col for col in X_train.columns if X_train[col].dtypes == 'object']\n\ncategorical_cols = [col for col in categorical_cols if col not in ['KitchenQual', 'HeatingQC', 'ExterCond', 'ExterQual']]\n\ntarget_encoder = MEstimateEncoder(cols = X_train[categorical_cols], m=5.0)\n\nX_train_target = target_encoder.fit_transform(X_train, y_train)\n\nX_test_target = target_encoder.transform(X_test)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Feature Engineering**","metadata":{}},{"cell_type":"markdown","source":"Creating a total living space feature.","metadata":{}},{"cell_type":"code","source":"X_train_target['Total_Living'] = X_train_target['GrLivArea'] + X_train_target['TotalBsmtSF'] + X_train_target['GarageArea']\n\nX_test_target['Total_Living'] = X_test_target['GrLivArea'] + X_test_target['TotalBsmtSF'] + X_test_target['GarageArea']","metadata":{"execution":{"iopub.status.busy":"2022-07-08T16:09:20.335559Z","iopub.execute_input":"2022-07-08T16:09:20.335914Z","iopub.status.idle":"2022-07-08T16:09:20.343829Z","shell.execute_reply.started":"2022-07-08T16:09:20.335886Z","shell.execute_reply":"2022-07-08T16:09:20.342622Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Hyperparameter Tuning for XGBoost**","metadata":{}},{"cell_type":"markdown","source":"Here we are running a GridSearchCV to tune the parameters for XGBoost. This step will take several minutes. I am commenting this out in my notebook because I have already run it once. You can run it to see the best parameters for your dataset and model.","metadata":{}},{"cell_type":"code","source":"# # Hyperparameter Tuning for XGBoost\n\n# parameters = {'n_estimators':[500, 750, 1000], 'learning_rate':[0.02, 0.05], 'max_depth':[6, 8], 'subsample':[0.3, 0.5, 0.7], }\n\n# xgb = XGBRegressor(objective='reg:squarederror')\n\n# grid = GridSearchCV(xgb, parameters)\n\n# grid.fit(X_train_target, y_train_log)\n\n# print(grid.best_params_)","metadata":{"execution":{"iopub.status.busy":"2022-07-08T16:09:21.516058Z","iopub.execute_input":"2022-07-08T16:09:21.516496Z","iopub.status.idle":"2022-07-08T16:09:21.521101Z","shell.execute_reply.started":"2022-07-08T16:09:21.516456Z","shell.execute_reply":"2022-07-08T16:09:21.519914Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Standard Scaling**","metadata":{}},{"cell_type":"code","source":"numerical_cols = [col for col in X_train_target.columns if X_train_target[col].dtypes != 'object']\n\nscaler = StandardScaler()\nX_train_target_scaled = scaler.fit_transform(X_train_target[numerical_cols])\nX_test_target_scaled = scaler.transform(X_test_target[numerical_cols])\n\nX_train_target[numerical_cols] = X_train_target_scaled\nX_test_target[numerical_cols] = X_test_target_scaled","metadata":{"execution":{"iopub.status.busy":"2022-07-08T16:09:22.537663Z","iopub.execute_input":"2022-07-08T16:09:22.53832Z","iopub.status.idle":"2022-07-08T16:09:22.561302Z","shell.execute_reply.started":"2022-07-08T16:09:22.538285Z","shell.execute_reply":"2022-07-08T16:09:22.560505Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**NOTE:** XGBoost does not require the standardization of features, but other models do. Therefore I am scaling the data.","metadata":{}},{"cell_type":"markdown","source":"# **Modeling**","metadata":{}},{"cell_type":"markdown","source":"**NOTE:** I will run the model for both training and test data to see if there is overfitting.","metadata":{}},{"cell_type":"code","source":"# Modeling with XGBoost\n\nxgb = XGBRegressor(objective='reg:squarederror',n_estimators=1000, learning_rate=0.02, max_depth=6, subsample=0.5, colsample_bytree=0.5, min_child_weight=0.1)\n\nprint(\"Modeling log transformation\")\nrun_model(xgb, X_train_target, X_test_target, y_train_log, y_test_log)","metadata":{"execution":{"iopub.status.busy":"2022-07-08T16:10:29.11332Z","iopub.execute_input":"2022-07-08T16:10:29.113761Z","iopub.status.idle":"2022-07-08T16:10:33.657718Z","shell.execute_reply.started":"2022-07-08T16:10:29.113727Z","shell.execute_reply":"2022-07-08T16:10:33.656849Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We can clearly see the model is overfit. In order to overcome this issue, we will stack some other models on top of each other. I am using Yen Wee Lim's approach in stacking models. You can find their notebook in the following link.I am using Yen Wee Lim's approach in stacking models. You can find their notebook in the following link.","metadata":{}},{"cell_type":"markdown","source":"[YEN WEE LIM - Stacked Ensemble Models (Top 3% on Leaderboard)](https://www.kaggle.com/code/limyenwee/stacked-ensemble-models-top-3-on-leaderboard/notebook)","metadata":{}},{"cell_type":"markdown","source":"**Stacking Models**","metadata":{}},{"cell_type":"code","source":"ridgemodel = Ridge(alpha=26)\n\nxgbmodel = XGBRegressor(objective='reg:squarederror', n_estimators=1000, learning_rate=0.02)\n\nsvrmodel = SVR(C=8, epsilon=0.00005, gamma=0.0008)\n\nhubermodel = HuberRegressor(alpha=30,epsilon=3,fit_intercept=True,max_iter=2000)\n\ncbmodel = cb.CatBoostRegressor(loss_function='RMSE',colsample_bylevel=0.3, depth=2, \\\n          l2_leaf_reg=20, learning_rate=0.005, n_estimators=15000, subsample=0.3,verbose=False)\n\nestimators = [('ridgemodel', Ridge(alpha=26)), ('svrmodel', SVR(C=8, epsilon=0.00005, gamma=0.0008)), ('hubermodel', HuberRegressor(alpha=30,epsilon=3,fit_intercept=True,max_iter=10000)), ('cbmodel', cb.CatBoostRegressor(loss_function='RMSE',colsample_bylevel=0.3, depth=2, \\\n          l2_leaf_reg=20, learning_rate=0.005, n_estimators=15000, subsample=0.3,verbose=False))]\n\nstackmodel = StackingRegressor(estimators=estimators, final_estimator=XGBRegressor(objective='reg:squarederror', n_estimators=1000, learning_rate=0.02))","metadata":{"execution":{"iopub.status.busy":"2022-07-08T16:16:55.258497Z","iopub.execute_input":"2022-07-08T16:16:55.258906Z","iopub.status.idle":"2022-07-08T16:16:55.274638Z","shell.execute_reply.started":"2022-07-08T16:16:55.258876Z","shell.execute_reply":"2022-07-08T16:16:55.273217Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"-----------------------------\")\nprint(\"Overview of model performance\")\nprint(\"-----------------------------\")\nfor i in [ridgemodel,hubermodel,cbmodel,svrmodel,xgbmodel,stackmodel]:\n    i.fit(X_train_target, y_train_log)\n\n    print(f\"For model {i}\")\n    print(f\"Train RMSE: {round(math.sqrt(mse(y_train_log,i.predict(X_train_target))), 4)}\")\n    print(f\"Test RMSE: {round(math.sqrt(mse(y_test_log,i.predict(X_test_target))), 4)}\")\n    print(\"-----------------------------\")\nprint(\"-----------------------------\")\nprint(\"Average Test RMSE for ensemble model: \")\nfit = (svrmodel.predict(X_test_target) + xgbmodel.predict(X_test_target) + stackmodel.predict(X_test_target) + ridgemodel.predict(X_test_target) + hubermodel.predict(X_test_target) + cbmodel.predict(X_test_target)) / 6\nprint(math.sqrt(mse(y_test_log,fit)))","metadata":{"execution":{"iopub.status.busy":"2022-07-08T16:17:19.655246Z","iopub.execute_input":"2022-07-08T16:17:19.655649Z","iopub.status.idle":"2022-07-08T16:18:17.411335Z","shell.execute_reply.started":"2022-07-08T16:17:19.655616Z","shell.execute_reply":"2022-07-08T16:18:17.410068Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"final_prediction = (ridgemodel.predict(X_test_target) + 3 * xgbmodel.predict(X_test_target) \\\n+  5 * stackmodel.predict(X_test_target) + 4 * svrmodel.predict(X_test_target) \\\n+  hubermodel.predict(X_test_target) +  cbmodel.predict(X_test_target)) / 15\n\nround(math.sqrt(mse(y_test_log,final_prediction)), 4)","metadata":{"execution":{"iopub.status.busy":"2022-07-08T16:19:24.928196Z","iopub.execute_input":"2022-07-08T16:19:24.92859Z","iopub.status.idle":"2022-07-08T16:19:25.194878Z","shell.execute_reply.started":"2022-07-08T16:19:24.928556Z","shell.execute_reply":"2022-07-08T16:19:25.193607Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Making a Submission**","metadata":{}},{"cell_type":"markdown","source":"Here, we apply the exact same steps that we applied in the modeling section.","metadata":{}},{"cell_type":"markdown","source":"**Handling Missing Values**","metadata":{}},{"cell_type":"code","source":"X_train = train_df.drop(columns=['SalePrice', 'Id'])\ny_train = train_df['SalePrice']\nX_test = test_df.drop(columns=['Id'])\n\nnumerical_cols = [col for col in X_train.columns if X_train[col].dtypes != 'object']\ncategorical_cols = [col for col in X_train.columns if X_train[col].dtypes == 'object']\n\nnumerical_na_cols = []\ncategorical_na_cols = []\n\nfor k,v in X_train[numerical_cols].isnull().sum().to_dict().items():\n  if v != 0:\n    numerical_na_cols.append(k)\n\nfor k,v in X_train[categorical_cols].isnull().sum().to_dict().items():\n  if v != 0:\n    categorical_na_cols.append(k)\n\nprint(numerical_na_cols)\nprint(categorical_na_cols)","metadata":{"execution":{"iopub.status.busy":"2022-07-08T16:19:56.957268Z","iopub.execute_input":"2022-07-08T16:19:56.957697Z","iopub.status.idle":"2022-07-08T16:19:56.98378Z","shell.execute_reply.started":"2022-07-08T16:19:56.957661Z","shell.execute_reply":"2022-07-08T16:19:56.982793Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Imputing numerical columns with missing data \n\nmy_imputer = SimpleImputer(strategy='mean')\n\ntrain_imputed_values = my_imputer.fit_transform(X_train[numerical_na_cols])\ntest_imputed_values = my_imputer.transform(X_test[numerical_na_cols])\n\nX_train[numerical_na_cols] = train_imputed_values\nX_test[numerical_na_cols] = test_imputed_values","metadata":{"execution":{"iopub.status.busy":"2022-07-08T16:20:03.495355Z","iopub.execute_input":"2022-07-08T16:20:03.495781Z","iopub.status.idle":"2022-07-08T16:20:03.510139Z","shell.execute_reply.started":"2022-07-08T16:20:03.495749Z","shell.execute_reply":"2022-07-08T16:20:03.509163Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"values = {\"Alley\": 'No Alley Access', \"PoolQC\": \"No Pool\", \"Fence\": \"No Fence\", \"MiscFeature\": \"None\"}\n\nX_train.fillna(value=values, inplace=True)\nX_test.fillna(value=values, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-08T16:20:11.128589Z","iopub.execute_input":"2022-07-08T16:20:11.128961Z","iopub.status.idle":"2022-07-08T16:20:11.138709Z","shell.execute_reply.started":"2022-07-08T16:20:11.128931Z","shell.execute_reply":"2022-07-08T16:20:11.137553Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"drop_cols = [col for col in X_train.columns if X_train[col].isnull().any()]\n\nX_train = X_train.drop(columns=drop_cols)\nX_test = X_test.drop(columns=drop_cols)","metadata":{"execution":{"iopub.status.busy":"2022-07-08T16:20:19.120001Z","iopub.execute_input":"2022-07-08T16:20:19.120382Z","iopub.status.idle":"2022-07-08T16:20:19.150253Z","shell.execute_reply.started":"2022-07-08T16:20:19.120351Z","shell.execute_reply":"2022-07-08T16:20:19.148938Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Checking columns with missing values\n\nprint('Train columns with missing data:')\n\nfor k,v in X_train.isnull().sum().to_dict().items():\n  if v != 0:\n    print(f\"{k}:{v}\")\n  else:\n    continue\n\nprint('---------------------------------')\nprint('Test columns with missing data')\n\ntest_missing_cols = []\nfor k,v in X_test.isnull().sum().to_dict().items():\n  if v != 0:\n    print(f\"{k}:{v}\")\n    test_missing_cols.append(k)\n  else:\n    continue","metadata":{"execution":{"iopub.status.busy":"2022-07-08T16:20:30.973835Z","iopub.execute_input":"2022-07-08T16:20:30.974232Z","iopub.status.idle":"2022-07-08T16:20:30.991893Z","shell.execute_reply.started":"2022-07-08T16:20:30.974203Z","shell.execute_reply":"2022-07-08T16:20:30.990566Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Test Data still has some missing values. Will deal with them using imputing on test data.","metadata":{}},{"cell_type":"code","source":"num_test_missing = [col for col in test_missing_cols if X_test[col].dtypes != 'object']\ncat_test_missing  = [col for col in test_missing_cols if X_test[col].dtypes == 'object']\n\nmy_imputer = SimpleImputer(strategy='mean')\n\ntest_imputed_values = my_imputer.fit_transform(X_test[num_test_missing])\n\nX_test[num_test_missing] = test_imputed_values","metadata":{"execution":{"iopub.status.busy":"2022-07-08T16:20:56.761771Z","iopub.execute_input":"2022-07-08T16:20:56.762132Z","iopub.status.idle":"2022-07-08T16:20:56.775256Z","shell.execute_reply.started":"2022-07-08T16:20:56.762103Z","shell.execute_reply":"2022-07-08T16:20:56.774322Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"my_imputer_cat = SimpleImputer(strategy='most_frequent')\n\ntest_imputed_values = my_imputer_cat.fit_transform(X_test[cat_test_missing])\n\nX_test[cat_test_missing] = test_imputed_values","metadata":{"execution":{"iopub.status.busy":"2022-07-08T16:21:17.037336Z","iopub.execute_input":"2022-07-08T16:21:17.037746Z","iopub.status.idle":"2022-07-08T16:21:17.050014Z","shell.execute_reply.started":"2022-07-08T16:21:17.037716Z","shell.execute_reply":"2022-07-08T16:21:17.049043Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Target Encoding**","metadata":{}},{"cell_type":"code","source":"y_train_log = np.log10(y_train)\n\ncategorical_cols = [col for col in X_train.columns if X_train[col].dtypes == 'object']\n\ntarget_encoder = MEstimateEncoder(cols = X_train[categorical_cols], m=5.0)\n\nX_train_target = target_encoder.fit_transform(X_train, y_train)\n\nX_test_target = target_encoder.transform(X_test)","metadata":{"execution":{"iopub.status.busy":"2022-07-08T16:21:37.836488Z","iopub.execute_input":"2022-07-08T16:21:37.836842Z","iopub.status.idle":"2022-07-08T16:21:38.277581Z","shell.execute_reply.started":"2022-07-08T16:21:37.836813Z","shell.execute_reply":"2022-07-08T16:21:38.276348Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Feature Engineering**","metadata":{}},{"cell_type":"code","source":"X_train_target['Total_Living'] = X_train_target['GrLivArea'] + X_train_target['TotalBsmtSF'] + X_train_target['GarageArea']\nX_test_target['Total_Living'] = X_test_target['GrLivArea'] + X_test_target['TotalBsmtSF'] + X_test_target['GarageArea']","metadata":{"execution":{"iopub.status.busy":"2022-07-08T16:21:50.278392Z","iopub.execute_input":"2022-07-08T16:21:50.278826Z","iopub.status.idle":"2022-07-08T16:21:50.287223Z","shell.execute_reply.started":"2022-07-08T16:21:50.278793Z","shell.execute_reply":"2022-07-08T16:21:50.286365Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Standard Scaling**","metadata":{}},{"cell_type":"code","source":"numerical_cols = [col for col in X_train_target.columns if X_train_target[col].dtypes != 'object']\n\nscaler = StandardScaler()\nX_train_target_scaled = scaler.fit_transform(X_train_target[numerical_cols])\nX_test_target_scaled = scaler.transform(X_test_target[numerical_cols])\n\nX_train_target[numerical_cols] = X_train_target_scaled\nX_test_target[numerical_cols] = X_test_target_scaled","metadata":{"execution":{"iopub.status.busy":"2022-07-08T16:22:05.386278Z","iopub.execute_input":"2022-07-08T16:22:05.386719Z","iopub.status.idle":"2022-07-08T16:22:05.423538Z","shell.execute_reply.started":"2022-07-08T16:22:05.386684Z","shell.execute_reply":"2022-07-08T16:22:05.422314Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Modeling**","metadata":{}},{"cell_type":"code","source":"ridgemodel = Ridge(alpha=26)\n\n# xgbmodel = XGBRegressor(alpha= 3, colsample_bytree=0.5, reg_lambda=3, learning_rate= 0.01,\\\n#            max_depth=3, n_estimators=10000, subsample=0.65)\n\nxgbmodel = XGBRegressor(objective='reg:squarederror', n_estimators=1000, learning_rate=0.02)\n\nsvrmodel = SVR(C=8, epsilon=0.00005, gamma=0.0008)\n\nhubermodel = HuberRegressor(alpha=30,epsilon=3,fit_intercept=True,max_iter=2000)\n\ncbmodel = cb.CatBoostRegressor(loss_function='RMSE',colsample_bylevel=0.3, depth=2, \\\n          l2_leaf_reg=20, learning_rate=0.005, n_estimators=15000, subsample=0.3,verbose=False)\n\nestimators = [('ridgemodel', Ridge(alpha=26)), ('svrmodel', SVR(C=8, epsilon=0.00005, gamma=0.0008)), ('hubermodel', HuberRegressor(alpha=30,epsilon=3,fit_intercept=True,max_iter=10000)), ('cbmodel', cb.CatBoostRegressor(loss_function='RMSE',colsample_bylevel=0.3, depth=2, \\\n          l2_leaf_reg=20, learning_rate=0.005, n_estimators=15000, subsample=0.3,verbose=False))]\n\nstackmodel = StackingRegressor(estimators=estimators, final_estimator=XGBRegressor(objective='reg:squarederror', n_estimators=1000, learning_rate=0.02))","metadata":{"execution":{"iopub.status.busy":"2022-07-08T16:22:39.065585Z","iopub.execute_input":"2022-07-08T16:22:39.065973Z","iopub.status.idle":"2022-07-08T16:22:39.081577Z","shell.execute_reply.started":"2022-07-08T16:22:39.06594Z","shell.execute_reply":"2022-07-08T16:22:39.080174Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ridgemodel.fit(X_train_target, y_train_log)\nxgbmodel.fit(X_train_target, y_train_log)\nhubermodel.fit(X_train_target, y_train_log)\nsvrmodel.fit(X_train_target, y_train_log)\ncbmodel.fit(X_train_target, y_train_log)\nstackmodel.fit(X_train_target, y_train_log)\n\nfinal_prediction = (ridgemodel.predict(X_test_target) + 3 * xgbmodel.predict(X_test_target) \\\n+  5 * stackmodel.predict(X_test_target) + 4 * svrmodel.predict(X_test_target) \\\n+  hubermodel.predict(X_test_target) +  cbmodel.predict(X_test_target)) / 15","metadata":{"execution":{"iopub.status.busy":"2022-07-08T16:23:24.483488Z","iopub.execute_input":"2022-07-08T16:23:24.483872Z","iopub.status.idle":"2022-07-08T16:24:24.521098Z","shell.execute_reply.started":"2022-07-08T16:23:24.483842Z","shell.execute_reply":"2022-07-08T16:24:24.519761Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Submission**","metadata":{}},{"cell_type":"code","source":"inverse_final_pred = 10 ** final_prediction\n\nsubmission = pd.DataFrame({'Id': test_df.Id, 'SalePrice': inverse_final_pred})\n\nsubmission.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-08T16:24:28.079228Z","iopub.execute_input":"2022-07-08T16:24:28.080231Z","iopub.status.idle":"2022-07-08T16:24:28.092999Z","shell.execute_reply.started":"2022-07-08T16:24:28.080191Z","shell.execute_reply":"2022-07-08T16:24:28.091903Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Please let me know what you think, especially if you see any problems or anything to improve on. Any feedback is appreciated. ","metadata":{}}]}