{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# House Price Prediction (Regression task)\n\nPreprocessing steps:\n- drop Id Column\n- feature Engineering\n- fill NA entries\n- normalize positively skewed numerical data\n","metadata":{}},{"cell_type":"markdown","source":"# Exploring the Data","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nimport statsmodels.api as sm\nimport scipy.stats as stats\nfrom scipy.stats import norm, skew\nfrom sklearn.impute import SimpleImputer, KNNImputer\nfrom sklearn.preprocessing import OrdinalEncoder, OneHotEncoder\nfrom sklearn.preprocessing import StandardScaler, normalize, PowerTransformer\n\n%matplotlib inline","metadata":{"execution":{"iopub.status.busy":"2022-07-14T15:32:25.033818Z","iopub.execute_input":"2022-07-14T15:32:25.034618Z","iopub.status.idle":"2022-07-14T15:32:27.211743Z","shell.execute_reply.started":"2022-07-14T15:32:25.034468Z","shell.execute_reply":"2022-07-14T15:32:27.210587Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n\ndf = pd.read_csv('../input/house-prices-advanced-regression-techniques/train.csv')\ndf_test = pd.read_csv('../input/house-prices-advanced-regression-techniques/test.csv')\n\n","metadata":{"execution":{"iopub.status.busy":"2022-07-14T15:32:27.213915Z","iopub.execute_input":"2022-07-14T15:32:27.214358Z","iopub.status.idle":"2022-07-14T15:32:27.304086Z","shell.execute_reply.started":"2022-07-14T15:32:27.214312Z","shell.execute_reply":"2022-07-14T15:32:27.303165Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"numerical_cols = [x for x in df.columns if df[x].dtype in('int64', 'float64')]\ncategorical_cols = [x for x in df.columns if df[x].dtype not in('int64', 'float64')]\n\ndf_numerical = df[numerical_cols]\nplt.figure(figsize=(20,15))\n\ncorrelation_mat = df_numerical.corr()\n# drop Id\ndf.drop(['Id'], axis=1, inplace=True)\ntargets = np.log1p(df['SalePrice'])\ndf.drop(['SalePrice'], axis=1, inplace=True)\ntest_ids = df_test['Id']\ndf_test.drop(['Id'], axis=1, inplace=True)\nprint(test_ids)\n\n\nmask = np.triu(np.ones_like(correlation_mat, dtype=bool))\ncorrelation_mat[(correlation_mat < 0.3) & (correlation_mat > -0.3)] = 0\nsns.heatmap(correlation_mat, annot=True, mask=mask, linewidth=0.5, cmap='Blues', vmin=0, vmax=1)","metadata":{"execution":{"iopub.status.busy":"2022-07-14T15:32:27.308377Z","iopub.execute_input":"2022-07-14T15:32:27.308770Z","iopub.status.idle":"2022-07-14T15:32:30.683957Z","shell.execute_reply.started":"2022-07-14T15:32:27.308733Z","shell.execute_reply":"2022-07-14T15:32:30.683000Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#create additional features\ndf['TotalSF'] = df['TotalBsmtSF'] + df['1stFlrSF'] + df['2ndFlrSF']\ndf_test['TotalSF'] = df_test['TotalBsmtSF'] + df_test['1stFlrSF'] + df_test['2ndFlrSF']\n\n\n#df['UnfinishedBsmntSFRatio'] = df['BsmtUnfSF'] / df['TotalBsmtSF']\n#df_test['UnfinishedBsmntSFRatio'] = df_test['BsmtUnfSF'] / df_test['TotalBsmtSF']\n\ndf.drop(['TotalBsmtSF', '1stFlrSF', '2ndFlrSF'], axis=1, inplace=True)\ndf_test.drop(['TotalBsmtSF', '1stFlrSF', '2ndFlrSF'], axis=1, inplace=True)\n\n#df['BulidYear'] = np.sqrt((df['YearBuilt'] * df['YearRemodAdd']).to_numpy())\n#df_test['BulidYear'] = np.sqrt((df_test['YearBuilt'] * df_test['YearRemodAdd']).to_numpy())\n\n#df.drop(['YearBuilt', 'YearRemodAdd'], axis=1, inplace=True)\n#df_test.drop(['YearBuilt', 'YearRemodAdd'], axis=1, inplace=True)\n\n#df['AreaPerRoomAbvGrd'] = df['GrLivArea'] / df['TotRmsAbvGrd']\n#df_test['AreaPerRoomAbvGrd'] = df_test['GrLivArea'] / df_test['TotRmsAbvGrd']\n\n#df['BulidYear'] = np.sqrt((df['YearBuilt'] * df['YearRemodAdd']).to_numpy())\n#df_test['BulidYear'] = np.sqrt((df_test['YearBuilt'] * df_test['YearRemodAdd']).to_numpy())\n\n\n\n#df['Total_Home_Quality'] = df['OverallQual'] + df['OverallCond']\n#df_test['Total_Home_Quality'] = df_test['OverallQual'] + df_test['OverallCond']\n","metadata":{"execution":{"iopub.status.busy":"2022-07-14T15:32:30.686432Z","iopub.execute_input":"2022-07-14T15:32:30.687035Z","iopub.status.idle":"2022-07-14T15:32:30.701692Z","shell.execute_reply.started":"2022-07-14T15:32:30.687002Z","shell.execute_reply":"2022-07-14T15:32:30.700326Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for col in df.isnull().keys():\n    # fill with most common entry\n    df[col]=df[col].fillna(df[col].mode()[0])\n    \nfor col in df_test.isnull().keys():\n    # fill with most common entry\n    df_test[col]=df_test[col].fillna(df_test[col].mode()[0])  ","metadata":{"execution":{"iopub.status.busy":"2022-07-14T15:32:30.703269Z","iopub.execute_input":"2022-07-14T15:32:30.703641Z","iopub.status.idle":"2022-07-14T15:32:30.847430Z","shell.execute_reply.started":"2022-07-14T15:32:30.703608Z","shell.execute_reply":"2022-07-14T15:32:30.846576Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# dummy variables from categorical features\ndf1 = df\ndf_test1 = df_test\ntrain_test = pd.concat([df,df_test], axis=0, sort=False)\ntrain_test.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-14T15:32:30.849027Z","iopub.execute_input":"2022-07-14T15:32:30.849783Z","iopub.status.idle":"2022-07-14T15:32:30.875302Z","shell.execute_reply.started":"2022-07-14T15:32:30.849732Z","shell.execute_reply":"2022-07-14T15:32:30.874527Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# GrLivArea before transformation\n\n#in the right chart we can see that the data is squeezed to the left side -> this is called it is right skewed\n# (or positively skewed, as the skew values are positive)\n\nfig, ax = plt.subplots(1,2, figsize= (15,5))\nfig.suptitle(\"Prior normalization: qq-plot & distribution GrLivArea \", fontsize= 15)\n\nsm.qqplot(df1[\"GrLivArea\"], stats.t, distargs=(4,),fit=True, line=\"45\", ax = ax[0])\n\nsns.distplot(df1[\"GrLivArea\"], kde = True, hist=True, fit = norm, ax = ax[1])\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-14T15:32:30.876584Z","iopub.execute_input":"2022-07-14T15:32:30.877089Z","iopub.status.idle":"2022-07-14T15:32:31.405006Z","shell.execute_reply.started":"2022-07-14T15:32:30.877057Z","shell.execute_reply":"2022-07-14T15:32:31.403728Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#get skewed features\nnumeric_features = train_test.dtypes[train_test.dtypes != object].index\nskewed_features = train_test[numeric_features].apply(lambda x: skew(x)).sort_values(ascending=False)\nhigh_skew = skewed_features[skewed_features > 0.5]\nskew_index = high_skew.index\n#normalize skewed features\nfor i in skew_index:\n    train_test[i] = np.log1p(train_test[i])\n#log1p = log(1 + x) -> we use this, as it is possible that we have 0 as a value -> log(0) not defined\n# and log(1) = 0 so if x = 0, log(1 + x) results in 0\n","metadata":{"execution":{"iopub.status.busy":"2022-07-14T15:32:31.406557Z","iopub.execute_input":"2022-07-14T15:32:31.406912Z","iopub.status.idle":"2022-07-14T15:32:31.440972Z","shell.execute_reply.started":"2022-07-14T15:32:31.406879Z","shell.execute_reply":"2022-07-14T15:32:31.440001Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# GrLivArea after transformation\n\n#after normalizing the data with the log we can see that the distribution now looks like a normal distribution and \n# the qq-plot supports that finding\n\nfig, ax = plt.subplots(1,2, figsize= (15,5))\nfig.suptitle(\"After normalization:  qq-plot & distribution GrLivArea \", fontsize= 15)\n\nsm.qqplot(train_test[\"GrLivArea\"], stats.t, distargs=(4,),fit=True, line=\"45\", ax = ax[0])\n\nsns.distplot(train_test[\"GrLivArea\"], kde = True, hist=True, fit = norm, ax = ax[1])\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-14T15:32:31.442558Z","iopub.execute_input":"2022-07-14T15:32:31.443705Z","iopub.status.idle":"2022-07-14T15:32:31.988789Z","shell.execute_reply.started":"2022-07-14T15:32:31.443656Z","shell.execute_reply":"2022-07-14T15:32:31.987664Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#encoding of categorical data\ntrain_test1 = pd.get_dummies(train_test)\ntrain_test1.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-14T15:32:31.992297Z","iopub.execute_input":"2022-07-14T15:32:31.992663Z","iopub.status.idle":"2022-07-14T15:32:32.054111Z","shell.execute_reply.started":"2022-07-14T15:32:31.992631Z","shell.execute_reply":"2022-07-14T15:32:32.052824Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_test1[1455:1465]","metadata":{"execution":{"iopub.status.busy":"2022-07-14T15:32:32.056025Z","iopub.execute_input":"2022-07-14T15:32:32.056490Z","iopub.status.idle":"2022-07-14T15:32:32.089316Z","shell.execute_reply.started":"2022-07-14T15:32:32.056443Z","shell.execute_reply":"2022-07-14T15:32:32.088190Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n\ndf1 = train_test1[:1460]\ndf1.head()\n","metadata":{"execution":{"iopub.status.busy":"2022-07-14T15:32:32.090862Z","iopub.execute_input":"2022-07-14T15:32:32.091198Z","iopub.status.idle":"2022-07-14T15:32:32.117640Z","shell.execute_reply.started":"2022-07-14T15:32:32.091168Z","shell.execute_reply":"2022-07-14T15:32:32.116539Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test1 = train_test1[1460:]\ndf_test1.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-14T15:32:32.119087Z","iopub.execute_input":"2022-07-14T15:32:32.119641Z","iopub.status.idle":"2022-07-14T15:32:32.143854Z","shell.execute_reply.started":"2022-07-14T15:32:32.119608Z","shell.execute_reply":"2022-07-14T15:32:32.142795Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Model Selection","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\n# splitting train data to train, validation\n# selecting columns\n# data = df[selected_columns]\n\n#targets = df1[['SalePrice']]\n#df1.drop(['SalePrice'], axis=1, inplace=True)\ndata = df1\n\n# splitting the data in training and validation set\nX_train, X_val, y_train, y_val = train_test_split(data, targets, test_size=0.10, random_state=123)\n#y_train = y_train.values\ny_val = y_val.values.ravel()\n\nX_train = X_train.values\ny_train = y_train.values.ravel()\n\n\nX_train.shape, y_train.shape # , X_val.shape, y_val.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-14T15:32:32.145762Z","iopub.execute_input":"2022-07-14T15:32:32.146100Z","iopub.status.idle":"2022-07-14T15:32:32.163973Z","shell.execute_reply.started":"2022-07-14T15:32:32.146058Z","shell.execute_reply":"2022-07-14T15:32:32.162961Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import mean_squared_error\nfrom sklearn.model_selection import KFold, cross_val_score\n\ndef RMSE(target, prediction):\n    return np.sqrt(mean_squared_error(target, prediction))\n\ndef CV_RMSE(model):\n    rmse = np.sqrt(-cross_val_score(model, X_train, y_train, scoring=\"neg_mean_squared_error\", cv=kf))\n    return (rmse)","metadata":{"execution":{"iopub.status.busy":"2022-07-14T15:32:32.164924Z","iopub.execute_input":"2022-07-14T15:32:32.165244Z","iopub.status.idle":"2022-07-14T15:32:32.172916Z","shell.execute_reply.started":"2022-07-14T15:32:32.165217Z","shell.execute_reply":"2022-07-14T15:32:32.171357Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from xgboost import XGBRegressor\nfrom catboost import Pool\nfrom sklearn.svm import SVR\nfrom lightgbm import LGBMRegressor\nfrom sklearn.naive_bayes import GaussianNB\nfrom sklearn.tree import DecisionTreeRegressor\nfrom mlxtend.regressor import StackingRegressor\nfrom sklearn.linear_model import LinearRegression, BayesianRidge\nfrom sklearn.model_selection import RepeatedKFold\nfrom sklearn.model_selection import KFold, cross_val_score\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.ensemble import GradientBoostingRegressor, RandomForestRegressor\nfrom sklearn.model_selection import GridSearchCV, RandomizedSearchCV\nfrom sklearn.metrics import mean_squared_error, mean_absolute_error, mean_squared_log_error\nfrom catboost import CatBoostRegressor\nimport catboost","metadata":{"execution":{"iopub.status.busy":"2022-07-14T15:32:32.174527Z","iopub.execute_input":"2022-07-14T15:32:32.175067Z","iopub.status.idle":"2022-07-14T15:32:33.506316Z","shell.execute_reply.started":"2022-07-14T15:32:32.175022Z","shell.execute_reply":"2022-07-14T15:32:33.505356Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 10 Fold Cross validation\n\nkf = KFold(n_splits=10, random_state=42, shuffle=True)\n\ncv_scores = []\ncv_std = []\n\nbaseline_models = ['Linear_Reg.','Bayesian_Ridge_Reg.','LGBM_Reg.','SVR',\n                   'Dec_Tree_Reg.','Random_Forest_Reg.', 'XGB_Reg.',\n                   'Grad_Boost_Reg.','Cat_Boost_Reg.','Stacked_Reg.']","metadata":{"execution":{"iopub.status.busy":"2022-07-14T15:32:33.508139Z","iopub.execute_input":"2022-07-14T15:32:33.508658Z","iopub.status.idle":"2022-07-14T15:32:33.516275Z","shell.execute_reply.started":"2022-07-14T15:32:33.508607Z","shell.execute_reply":"2022-07-14T15:32:33.515036Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Conclusions","metadata":{}},{"cell_type":"code","source":"params = {'iterations': 20000,\n          'learning_rate': 0.005,\n          'depth': 4,\n          'l2_leaf_reg': 1,\n          'eval_metric':'RMSE',\n          'early_stopping_rounds': 1000,\n          'verbose': 200,\n          'random_seed': 42}\ncat_f = CatBoostRegressor(**params)\ncat_f.fit(X_train,y_train,\n                        eval_set = (X_val,y_val),\n                     plot=False,\n                     verbose = False)","metadata":{"execution":{"iopub.status.busy":"2022-07-14T15:32:33.518694Z","iopub.execute_input":"2022-07-14T15:32:33.519166Z","iopub.status.idle":"2022-07-14T15:32:54.343643Z","shell.execute_reply.started":"2022-07-14T15:32:33.519120Z","shell.execute_reply":"2022-07-14T15:32:54.342576Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_pred = cat_f.predict(df_test1)\nsubmission = pd.DataFrame(test_ids, columns = ['Id'])\ntest_pred = np.expm1(test_pred)\nsubmission['SalePrice'] = test_pred \nsubmission.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-14T15:32:54.344977Z","iopub.execute_input":"2022-07-14T15:32:54.345318Z","iopub.status.idle":"2022-07-14T15:32:54.579518Z","shell.execute_reply.started":"2022-07-14T15:32:54.345287Z","shell.execute_reply":"2022-07-14T15:32:54.578581Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nsubmission.to_csv(\"submission.csv\", index = False, header = True)","metadata":{"execution":{"iopub.status.busy":"2022-07-14T15:32:54.580836Z","iopub.execute_input":"2022-07-14T15:32:54.581529Z","iopub.status.idle":"2022-07-14T15:32:54.597152Z","shell.execute_reply.started":"2022-07-14T15:32:54.581466Z","shell.execute_reply":"2022-07-14T15:32:54.596001Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}