{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-07T13:15:51.653513Z","iopub.execute_input":"2022-07-07T13:15:51.653871Z","iopub.status.idle":"2022-07-07T13:15:51.664274Z","shell.execute_reply.started":"2022-07-07T13:15:51.653837Z","shell.execute_reply":"2022-07-07T13:15:51.663161Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nimport plotly.express as px\nimport plotly.io as pio\nimport scipy.stats as st\nimport math\nimport missingno as msno\nfrom scipy.stats import norm, skew\nfrom collections import Counter\n%matplotlib inline\n\nfrom sklearn.preprocessing import LabelEncoder\nfrom sklearn.preprocessing import MinMaxScaler, RobustScaler, StandardScaler\nfrom sklearn.model_selection import train_test_split, GridSearchCV, cross_val_score, KFold\nfrom sklearn.metrics import mean_squared_error, mean_squared_log_error, r2_score\nfrom sklearn import model_selection\nfrom sklearn.pipeline import make_pipeline\n\nfrom xgboost import XGBRegressor\nfrom lightgbm import LGBMRegressor\nfrom sklearn.svm import SVR\nfrom sklearn.ensemble import RandomForestRegressor, GradientBoostingRegressor\nfrom sklearn.linear_model import Ridge, RidgeCV, Lasso, LassoCV\nfrom mlxtend.regressor import StackingCVRegressor\n\n# to ignore warnings\nimport warnings\nwarnings.filterwarnings(\"ignore\")\n\n#to see model hyperparameters\nfrom sklearn import set_config\nset_config(print_changed_only = False)\n\n# to show all columns\npd.set_option('display.max_columns', 82)","metadata":{"execution":{"iopub.status.busy":"2022-07-07T13:15:51.760022Z","iopub.execute_input":"2022-07-07T13:15:51.760614Z","iopub.status.idle":"2022-07-07T13:15:51.774971Z","shell.execute_reply.started":"2022-07-07T13:15:51.760580Z","shell.execute_reply":"2022-07-07T13:15:51.772913Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **EDA**","metadata":{}},{"cell_type":"code","source":"train = pd.read_csv('../input/house-prices-advanced-regression-techniques/train.csv')\ntest = pd.read_csv('../input/house-prices-advanced-regression-techniques/test.csv')","metadata":{"execution":{"iopub.status.busy":"2022-07-07T13:15:51.861855Z","iopub.execute_input":"2022-07-07T13:15:51.862113Z","iopub.status.idle":"2022-07-07T13:15:51.901823Z","shell.execute_reply.started":"2022-07-07T13:15:51.862090Z","shell.execute_reply":"2022-07-07T13:15:51.900962Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.shape, test.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-07T13:15:51.929095Z","iopub.execute_input":"2022-07-07T13:15:51.929361Z","iopub.status.idle":"2022-07-07T13:15:51.935324Z","shell.execute_reply.started":"2022-07-07T13:15:51.929337Z","shell.execute_reply":"2022-07-07T13:15:51.934310Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.select_dtypes(include=['object']).nunique()","metadata":{"execution":{"iopub.status.busy":"2022-07-07T13:15:51.996155Z","iopub.execute_input":"2022-07-07T13:15:51.997956Z","iopub.status.idle":"2022-07-07T13:15:52.016777Z","shell.execute_reply.started":"2022-07-07T13:15:51.997929Z","shell.execute_reply":"2022-07-07T13:15:52.015846Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-07T13:15:52.062373Z","iopub.execute_input":"2022-07-07T13:15:52.064116Z","iopub.status.idle":"2022-07-07T13:15:52.087744Z","shell.execute_reply.started":"2022-07-07T13:15:52.064091Z","shell.execute_reply":"2022-07-07T13:15:52.086710Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Target Variable**","metadata":{}},{"cell_type":"code","source":"target_skew = train['SalePrice'].skew()\nprint(target_skew)\nsns.distplot(train['SalePrice'])","metadata":{"execution":{"iopub.status.busy":"2022-07-07T13:15:52.173823Z","iopub.execute_input":"2022-07-07T13:15:52.174306Z","iopub.status.idle":"2022-07-07T13:15:52.448715Z","shell.execute_reply.started":"2022-07-07T13:15:52.174278Z","shell.execute_reply":"2022-07-07T13:15:52.447661Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"There is a postive skewness. So transformation should be applied to the skewed data","metadata":{}},{"cell_type":"code","source":"sns.distplot(np.log1p(train['SalePrice']), fit = norm)\nprint(np.log1p(train['SalePrice']).skew())","metadata":{"execution":{"iopub.status.busy":"2022-07-07T13:15:52.450729Z","iopub.execute_input":"2022-07-07T13:15:52.451090Z","iopub.status.idle":"2022-07-07T13:15:52.701910Z","shell.execute_reply.started":"2022-07-07T13:15:52.451054Z","shell.execute_reply":"2022-07-07T13:15:52.700738Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.distplot(np.log(train['SalePrice']), fit = norm)\nprint(np.log(train['SalePrice']).skew())","metadata":{"execution":{"iopub.status.busy":"2022-07-07T13:15:52.703815Z","iopub.execute_input":"2022-07-07T13:15:52.704450Z","iopub.status.idle":"2022-07-07T13:15:52.947913Z","shell.execute_reply.started":"2022-07-07T13:15:52.704413Z","shell.execute_reply":"2022-07-07T13:15:52.946946Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from scipy.stats import skew\nfrom scipy.special import boxcox\nsns.distplot(boxcox(train['SalePrice'], 0), fit = norm)\nprint(skew(boxcox(train['SalePrice'], 0)))","metadata":{"execution":{"iopub.status.busy":"2022-07-07T13:15:52.951167Z","iopub.execute_input":"2022-07-07T13:15:52.951546Z","iopub.status.idle":"2022-07-07T13:15:53.321746Z","shell.execute_reply.started":"2022-07-07T13:15:52.951509Z","shell.execute_reply":"2022-07-07T13:15:53.320838Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['SalePrice'] = np.log1p(train['SalePrice'])","metadata":{"execution":{"iopub.status.busy":"2022-07-07T13:15:53.323278Z","iopub.execute_input":"2022-07-07T13:15:53.323791Z","iopub.status.idle":"2022-07-07T13:15:53.329331Z","shell.execute_reply.started":"2022-07-07T13:15:53.323754Z","shell.execute_reply":"2022-07-07T13:15:53.328402Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['SalePrice'].skew()","metadata":{"execution":{"iopub.status.busy":"2022-07-07T13:15:53.330778Z","iopub.execute_input":"2022-07-07T13:15:53.331326Z","iopub.status.idle":"2022-07-07T13:15:53.362035Z","shell.execute_reply.started":"2022-07-07T13:15:53.331287Z","shell.execute_reply":"2022-07-07T13:15:53.360298Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Skewness is now fixed","metadata":{}},{"cell_type":"markdown","source":"**Feature Correlation**","metadata":{}},{"cell_type":"markdown","source":"How to find correlation between categorical features and SalePrice ??","metadata":{}},{"cell_type":"code","source":"corr = train.corr()\nhighest_corr_features = corr.index[abs(corr[\"SalePrice\"])>0.5]\nplt.figure(figsize=(20,20))\ng = sns.heatmap(train[highest_corr_features].corr(),annot=True,cmap=\"RdYlGn\")","metadata":{"execution":{"iopub.status.busy":"2022-07-07T13:15:53.363563Z","iopub.execute_input":"2022-07-07T13:15:53.364221Z","iopub.status.idle":"2022-07-07T13:15:54.849300Z","shell.execute_reply.started":"2022-07-07T13:15:53.364186Z","shell.execute_reply":"2022-07-07T13:15:54.848385Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We can find what are the features that are highly correlated with Saleprice and features that are correlated with each other","metadata":{}},{"cell_type":"code","source":"highest_corr_features","metadata":{"execution":{"iopub.status.busy":"2022-07-07T13:15:54.850537Z","iopub.execute_input":"2022-07-07T13:15:54.853074Z","iopub.status.idle":"2022-07-07T13:15:54.859971Z","shell.execute_reply.started":"2022-07-07T13:15:54.853033Z","shell.execute_reply":"2022-07-07T13:15:54.859023Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"corr[\"SalePrice\"].sort_values(ascending=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-07T13:15:54.861494Z","iopub.execute_input":"2022-07-07T13:15:54.862160Z","iopub.status.idle":"2022-07-07T13:15:54.876315Z","shell.execute_reply.started":"2022-07-07T13:15:54.862121Z","shell.execute_reply":"2022-07-07T13:15:54.875219Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Missing data**","metadata":{}},{"cell_type":"code","source":"y_train = train['SalePrice']\ntest_id = test['Id']\nall_data = pd.concat([train, test], axis=0, sort=False)\nall_data = all_data.drop(['Id', 'SalePrice'], axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-07-07T13:15:54.881941Z","iopub.execute_input":"2022-07-07T13:15:54.882224Z","iopub.status.idle":"2022-07-07T13:15:54.906932Z","shell.execute_reply.started":"2022-07-07T13:15:54.882199Z","shell.execute_reply":"2022-07-07T13:15:54.905860Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Total = all_data.isnull().sum().sort_values(ascending=False)\npercent = ((all_data.isnull().sum() / all_data.isnull().count())*100).sort_values(ascending=False)\nmissing_data = pd.concat([Total, percent], axis=1, keys=['Total', 'Percent'])\nmissing_data.head(18)","metadata":{"execution":{"iopub.status.busy":"2022-07-07T13:15:54.908730Z","iopub.execute_input":"2022-07-07T13:15:54.909413Z","iopub.status.idle":"2022-07-07T13:15:54.962533Z","shell.execute_reply.started":"2022-07-07T13:15:54.909378Z","shell.execute_reply":"2022-07-07T13:15:54.961560Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(missing_data.head(25).axes[0])\nprint(highest_corr_features)\n\nprint(\"High corr features with missing data - \" + highest_corr_features.intersection(missing_data.head(18).axes[0]))\n\nfeatures_to_remove = (missing_data.head(18).axes[0]).difference(highest_corr_features)\n\nprint(features_to_remove)","metadata":{"execution":{"iopub.status.busy":"2022-07-07T13:15:54.964021Z","iopub.execute_input":"2022-07-07T13:15:54.964464Z","iopub.status.idle":"2022-07-07T13:15:54.974302Z","shell.execute_reply.started":"2022-07-07T13:15:54.964424Z","shell.execute_reply":"2022-07-07T13:15:54.973327Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"High correlated features with missing data are MasVnrArea, GarageYrBlt. So removing features with no correaltion and lots of missing values","metadata":{}},{"cell_type":"code","source":"for j in features_to_remove:\n    all_data.drop(j, axis=1, inplace=True)\nall_data","metadata":{"execution":{"iopub.status.busy":"2022-07-07T13:15:54.975732Z","iopub.execute_input":"2022-07-07T13:15:54.976323Z","iopub.status.idle":"2022-07-07T13:15:55.063456Z","shell.execute_reply.started":"2022-07-07T13:15:54.976287Z","shell.execute_reply":"2022-07-07T13:15:55.062631Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Dropping Garageyrblt and Garage Area since they are highly correlated with yrblt and Garage cars","metadata":{}},{"cell_type":"code","source":"all_data.drop(['GarageYrBlt'], axis=1, inplace=True)\nall_data.drop(['GarageArea'], axis=1, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-07T13:15:55.064764Z","iopub.execute_input":"2022-07-07T13:15:55.065101Z","iopub.status.idle":"2022-07-07T13:15:55.073510Z","shell.execute_reply.started":"2022-07-07T13:15:55.065069Z","shell.execute_reply":"2022-07-07T13:15:55.072385Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_data.columns","metadata":{"execution":{"iopub.status.busy":"2022-07-07T13:15:55.075203Z","iopub.execute_input":"2022-07-07T13:15:55.075620Z","iopub.status.idle":"2022-07-07T13:15:55.086860Z","shell.execute_reply.started":"2022-07-07T13:15:55.075585Z","shell.execute_reply":"2022-07-07T13:15:55.086058Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"total = all_data.isnull().sum().sort_values(ascending=False)\ntotal.head(19)","metadata":{"execution":{"iopub.status.busy":"2022-07-07T13:15:55.088175Z","iopub.execute_input":"2022-07-07T13:15:55.088415Z","iopub.status.idle":"2022-07-07T13:15:55.111699Z","shell.execute_reply.started":"2022-07-07T13:15:55.088394Z","shell.execute_reply":"2022-07-07T13:15:55.110722Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Filling missing values**","metadata":{}},{"cell_type":"code","source":"all_data.describe()","metadata":{"execution":{"iopub.status.busy":"2022-07-07T13:15:55.112919Z","iopub.execute_input":"2022-07-07T13:15:55.113742Z","iopub.status.idle":"2022-07-07T13:15:55.206925Z","shell.execute_reply.started":"2022-07-07T13:15:55.113703Z","shell.execute_reply":"2022-07-07T13:15:55.205953Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# filling the numeric data\nnumeric_missed = ['BsmtFinSF1',\n                  'BsmtFinSF2',\n                  'BsmtUnfSF',\n                  'TotalBsmtSF',\n                  'BsmtFullBath',\n                  'BsmtHalfBath',\n                  'GarageCars']\n\nfor feature in numeric_missed:\n    all_data[feature] = all_data[feature].fillna(all_data[feature].mean())","metadata":{"execution":{"iopub.status.busy":"2022-07-07T13:15:55.208294Z","iopub.execute_input":"2022-07-07T13:15:55.208622Z","iopub.status.idle":"2022-07-07T13:15:55.219379Z","shell.execute_reply.started":"2022-07-07T13:15:55.208584Z","shell.execute_reply":"2022-07-07T13:15:55.218350Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#filling categorical data\ncategorical_missed = ['Exterior1st',\n                  'Exterior2nd',\n                  'SaleType',\n                  'MSZoning',\n                   'Electrical',\n                     'KitchenQual']\n\nfor feature in categorical_missed:\n    all_data[feature] = all_data[feature].fillna(all_data[feature].mode()[0])","metadata":{"execution":{"iopub.status.busy":"2022-07-07T13:15:55.220910Z","iopub.execute_input":"2022-07-07T13:15:55.221484Z","iopub.status.idle":"2022-07-07T13:15:55.238458Z","shell.execute_reply.started":"2022-07-07T13:15:55.221418Z","shell.execute_reply":"2022-07-07T13:15:55.237570Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_data['Functional'] = all_data['Functional'].fillna('Typ')\nall_data.drop(['Utilities'], axis=1, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-07T13:15:55.240906Z","iopub.execute_input":"2022-07-07T13:15:55.241633Z","iopub.status.idle":"2022-07-07T13:15:55.253267Z","shell.execute_reply.started":"2022-07-07T13:15:55.241596Z","shell.execute_reply":"2022-07-07T13:15:55.252253Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"total = all_data.isnull().sum().sort_values(ascending=False)\ntotal.head(5)","metadata":{"execution":{"iopub.status.busy":"2022-07-07T13:15:55.256442Z","iopub.execute_input":"2022-07-07T13:15:55.257485Z","iopub.status.idle":"2022-07-07T13:15:55.277642Z","shell.execute_reply.started":"2022-07-07T13:15:55.257457Z","shell.execute_reply":"2022-07-07T13:15:55.276642Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Fixing skew in other features**","metadata":{}},{"cell_type":"code","source":"numeric_feats = all_data.dtypes[all_data.dtypes != 'object'].index\nskewed_feats = all_data[numeric_feats].apply(lambda x: skew(x)).sort_values(ascending=False)\nhigh_skew = skewed_feats[abs(skewed_feats) > 0.5]\nhigh_skew","metadata":{"execution":{"iopub.status.busy":"2022-07-07T13:15:55.279094Z","iopub.execute_input":"2022-07-07T13:15:55.279585Z","iopub.status.idle":"2022-07-07T13:15:55.302599Z","shell.execute_reply.started":"2022-07-07T13:15:55.279531Z","shell.execute_reply":"2022-07-07T13:15:55.301777Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for feature in high_skew.index:\n    all_data[feature] = np.log1p(all_data[feature])","metadata":{"execution":{"iopub.status.busy":"2022-07-07T13:15:55.303598Z","iopub.execute_input":"2022-07-07T13:15:55.303862Z","iopub.status.idle":"2022-07-07T13:15:55.321756Z","shell.execute_reply.started":"2022-07-07T13:15:55.303837Z","shell.execute_reply":"2022-07-07T13:15:55.320881Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"numeric_feats = all_data.dtypes[all_data.dtypes != 'object'].index\nskewed_feats = all_data[numeric_feats].apply(lambda x: skew(x)).sort_values(ascending=False)\nhigh_skew = skewed_feats[abs(skewed_feats) > 0.5]\nhigh_skew","metadata":{"execution":{"iopub.status.busy":"2022-07-07T13:15:55.322937Z","iopub.execute_input":"2022-07-07T13:15:55.323343Z","iopub.status.idle":"2022-07-07T13:15:55.347394Z","shell.execute_reply.started":"2022-07-07T13:15:55.323309Z","shell.execute_reply":"2022-07-07T13:15:55.346465Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_data.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-07T13:15:55.348874Z","iopub.execute_input":"2022-07-07T13:15:55.349382Z","iopub.status.idle":"2022-07-07T13:15:55.399751Z","shell.execute_reply.started":"2022-07-07T13:15:55.349347Z","shell.execute_reply":"2022-07-07T13:15:55.398861Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Category to Numerical**","metadata":{}},{"cell_type":"code","source":"all_data = pd.get_dummies(all_data)\nall_data.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-07T13:15:55.401133Z","iopub.execute_input":"2022-07-07T13:15:55.401662Z","iopub.status.idle":"2022-07-07T13:15:55.480018Z","shell.execute_reply.started":"2022-07-07T13:15:55.401629Z","shell.execute_reply":"2022-07-07T13:15:55.478741Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_train =all_data[:len(y_train)]\nx_test = all_data[len(y_train):]","metadata":{"execution":{"iopub.status.busy":"2022-07-07T13:15:55.481634Z","iopub.execute_input":"2022-07-07T13:15:55.482230Z","iopub.status.idle":"2022-07-07T13:15:55.487047Z","shell.execute_reply.started":"2022-07-07T13:15:55.482187Z","shell.execute_reply":"2022-07-07T13:15:55.486050Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_test.shape , x_train.shape, y_train.shape\n","metadata":{"execution":{"iopub.status.busy":"2022-07-07T13:15:55.494736Z","iopub.execute_input":"2022-07-07T13:15:55.495476Z","iopub.status.idle":"2022-07-07T13:15:55.502720Z","shell.execute_reply.started":"2022-07-07T13:15:55.495439Z","shell.execute_reply":"2022-07-07T13:15:55.501598Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_train","metadata":{"execution":{"iopub.status.busy":"2022-07-07T13:15:55.504274Z","iopub.execute_input":"2022-07-07T13:15:55.504942Z","iopub.status.idle":"2022-07-07T13:15:55.570422Z","shell.execute_reply.started":"2022-07-07T13:15:55.504896Z","shell.execute_reply":"2022-07-07T13:15:55.569542Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**ML Models**","metadata":{}},{"cell_type":"code","source":"k_fold = KFold(n_splits = 15, random_state = 11, shuffle = True)","metadata":{"execution":{"iopub.status.busy":"2022-07-07T13:15:55.571648Z","iopub.execute_input":"2022-07-07T13:15:55.572534Z","iopub.status.idle":"2022-07-07T13:15:55.577288Z","shell.execute_reply.started":"2022-07-07T13:15:55.572499Z","shell.execute_reply":"2022-07-07T13:15:55.576418Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def cv_rmse(model):\n    rmse = np.sqrt(-cross_val_score(model, x_train, y_train, scoring = \"neg_mean_squared_error\", cv = k_fold))\n    return rmse\n\n\ndef rmsle(y, y_pred):\n    return np.sqrt(mean_squared_log_error(y, y_pred, squared = False))","metadata":{"execution":{"iopub.status.busy":"2022-07-07T13:15:55.578630Z","iopub.execute_input":"2022-07-07T13:15:55.579302Z","iopub.status.idle":"2022-07-07T13:15:55.588018Z","shell.execute_reply.started":"2022-07-07T13:15:55.579266Z","shell.execute_reply":"2022-07-07T13:15:55.586849Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#1 Linear regression","metadata":{}},{"cell_type":"code","source":"from sklearn import linear_model\n\nlasso = linear_model.Lasso(alpha = 0.5, max_iter = 1000000)","metadata":{"execution":{"iopub.status.busy":"2022-07-07T13:15:55.589288Z","iopub.execute_input":"2022-07-07T13:15:55.589782Z","iopub.status.idle":"2022-07-07T13:15:55.599020Z","shell.execute_reply.started":"2022-07-07T13:15:55.589746Z","shell.execute_reply":"2022-07-07T13:15:55.597960Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"score = cv_rmse(lasso)\nprint(score.mean())","metadata":{"execution":{"iopub.status.busy":"2022-07-07T13:15:55.600744Z","iopub.execute_input":"2022-07-07T13:15:55.601152Z","iopub.status.idle":"2022-07-07T13:15:56.071240Z","shell.execute_reply.started":"2022-07-07T13:15:55.601114Z","shell.execute_reply":"2022-07-07T13:15:56.069969Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ridge = linear_model.Ridge(alpha = 0.5, max_iter = 1000000)\nscore = cross_val_score(ridge, x_train, y_train)\nprint(score.mean())","metadata":{"execution":{"iopub.status.busy":"2022-07-07T13:15:56.072842Z","iopub.execute_input":"2022-07-07T13:15:56.073637Z","iopub.status.idle":"2022-07-07T13:15:56.293503Z","shell.execute_reply.started":"2022-07-07T13:15:56.073594Z","shell.execute_reply":"2022-07-07T13:15:56.292431Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_train.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-07T13:15:56.294999Z","iopub.execute_input":"2022-07-07T13:15:56.296649Z","iopub.status.idle":"2022-07-07T13:15:56.306899Z","shell.execute_reply.started":"2022-07-07T13:15:56.296611Z","shell.execute_reply":"2022-07-07T13:15:56.305790Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import Dense\nfrom tensorflow.keras.activations import linear, relu, sigmoid","metadata":{"execution":{"iopub.status.busy":"2022-07-07T13:15:56.309647Z","iopub.execute_input":"2022-07-07T13:15:56.311049Z","iopub.status.idle":"2022-07-07T13:15:56.319958Z","shell.execute_reply.started":"2022-07-07T13:15:56.311008Z","shell.execute_reply":"2022-07-07T13:15:56.318851Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = Sequential(\n    [               \n        ### START CODE HERE ###\n        tf.keras.Input(shape=(217,)),\n        Dense(units  = 217, activation = 'relu'),\n        Dense(units  = 150, activation = 'relu'),\n        Dense(units  = 75, activation = 'relu'),\n        Dense(units  = 25, activation = 'relu'),\n        Dense(units  = 5, activation = 'relu'),\n        Dense(units = 1, activation = 'linear')\n        ### END CODE HERE ### \n    ], name = \"my_model\" \n)\nmodel.compile(loss = 'mean_squared_error', optimizer=tf.keras.optimizers.Adam(learning_rate = 0.00001))","metadata":{"execution":{"iopub.status.busy":"2022-07-07T13:15:56.324674Z","iopub.execute_input":"2022-07-07T13:15:56.326429Z","iopub.status.idle":"2022-07-07T13:15:56.437965Z","shell.execute_reply.started":"2022-07-07T13:15:56.326383Z","shell.execute_reply":"2022-07-07T13:15:56.437016Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.summary()","metadata":{"execution":{"iopub.status.busy":"2022-07-07T13:15:56.439536Z","iopub.execute_input":"2022-07-07T13:15:56.439909Z","iopub.status.idle":"2022-07-07T13:15:56.448215Z","shell.execute_reply.started":"2022-07-07T13:15:56.439875Z","shell.execute_reply":"2022-07-07T13:15:56.446942Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = model.fit(x=x_train,y=y_train,\n          validation_split=0.05,\n          epochs=400)\nlosses = pd.DataFrame(model.history.history)\nlosses.plot()","metadata":{"execution":{"iopub.status.busy":"2022-07-07T13:15:56.449900Z","iopub.execute_input":"2022-07-07T13:15:56.450280Z","iopub.status.idle":"2022-07-07T13:17:18.976589Z","shell.execute_reply.started":"2022-07-07T13:15:56.450217Z","shell.execute_reply":"2022-07-07T13:17:18.975678Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"xgb = make_pipeline(RobustScaler(),\n                    XGBRegressor(colsample_bytree = 0.5, n_estimators = 6000,\n                                 max_depth = 4, learning_rate = 0.01, gamma = 0.45,\n                                 subsample = 0.5, random_state = 11, reg_alpha = 0.00006,\n                                 reg_lambda = None, nthread = -1))","metadata":{"execution":{"iopub.status.busy":"2022-07-07T13:17:18.977853Z","iopub.execute_input":"2022-07-07T13:17:18.978891Z","iopub.status.idle":"2022-07-07T13:17:18.984859Z","shell.execute_reply.started":"2022-07-07T13:17:18.978854Z","shell.execute_reply":"2022-07-07T13:17:18.983719Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# get CV score of the xgb model\nscore = cv_rmse(xgb)\nprint(\"Xgboost model's cross validation score: \", score.mean())","metadata":{"execution":{"iopub.status.busy":"2022-07-07T13:17:18.986399Z","iopub.execute_input":"2022-07-07T13:17:18.986747Z","iopub.status.idle":"2022-07-07T13:24:06.873342Z","shell.execute_reply.started":"2022-07-07T13:17:18.986714Z","shell.execute_reply":"2022-07-07T13:24:06.872491Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"lgbm = make_pipeline(RobustScaler(),\n                     LGBMRegressor(num_leaves = 6, bagging_fraction = 0.7,\n                                   bagging_freq = 4, min_sum_hessian_in_leaf = 11,\n                                   learning_rate = 0.01, n_estimators = 7500, max_bin = 200,\n                                   random_state = 11))","metadata":{"execution":{"iopub.status.busy":"2022-07-07T13:24:06.874655Z","iopub.execute_input":"2022-07-07T13:24:06.875354Z","iopub.status.idle":"2022-07-07T13:24:06.881917Z","shell.execute_reply.started":"2022-07-07T13:24:06.875318Z","shell.execute_reply":"2022-07-07T13:24:06.881201Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"score = cv_rmse(lgbm)\nprint(\"Light GBM model's cross validation score: \", score.mean())","metadata":{"execution":{"iopub.status.busy":"2022-07-07T13:24:06.883267Z","iopub.execute_input":"2022-07-07T13:24:06.884243Z","iopub.status.idle":"2022-07-07T13:25:10.063610Z","shell.execute_reply.started":"2022-07-07T13:24:06.884206Z","shell.execute_reply":"2022-07-07T13:25:10.062853Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gbr = make_pipeline(RobustScaler(),\n                    GradientBoostingRegressor(n_estimators = 7000, learning_rate = 0.01,\n                                              max_depth = 5, min_samples_split = 12, min_samples_leaf = 16,\n                                              loss = \"huber\", max_features = \"sqrt\", random_state = 11))","metadata":{"execution":{"iopub.status.busy":"2022-07-07T13:25:10.067072Z","iopub.execute_input":"2022-07-07T13:25:10.068569Z","iopub.status.idle":"2022-07-07T13:25:10.073924Z","shell.execute_reply.started":"2022-07-07T13:25:10.068532Z","shell.execute_reply":"2022-07-07T13:25:10.072749Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# get CV score of the gbr model\nscore = cv_rmse(gbr)\nprint(\"Gradient boosting model's cross validation score: \", score.mean())","metadata":{"execution":{"iopub.status.busy":"2022-07-07T13:25:10.075561Z","iopub.execute_input":"2022-07-07T13:25:10.076085Z","iopub.status.idle":"2022-07-07T13:35:13.670957Z","shell.execute_reply.started":"2022-07-07T13:25:10.076047Z","shell.execute_reply":"2022-07-07T13:35:13.669956Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rf = make_pipeline(RobustScaler(),\n                   RandomForestRegressor(n_estimators = 2500, max_depth = 15,\n                                         min_samples_split = 6, min_samples_leaf = 6,\n                                         random_state = 11))","metadata":{"execution":{"iopub.status.busy":"2022-07-07T13:35:13.672311Z","iopub.execute_input":"2022-07-07T13:35:13.672905Z","iopub.status.idle":"2022-07-07T13:35:13.679382Z","shell.execute_reply.started":"2022-07-07T13:35:13.672867Z","shell.execute_reply":"2022-07-07T13:35:13.678462Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# get CV score of the rf model\nscore = cv_rmse(rf)\nprint(\"Random forest model's cross validation score: \", score.mean())","metadata":{"execution":{"iopub.status.busy":"2022-07-07T13:35:13.680959Z","iopub.execute_input":"2022-07-07T13:35:13.681322Z","iopub.status.idle":"2022-07-07T13:42:24.440580Z","shell.execute_reply.started":"2022-07-07T13:35:13.681286Z","shell.execute_reply":"2022-07-07T13:42:24.439566Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"stacked = StackingCVRegressor(regressors = (xgb, lgbm, gbr, rf),\n                              meta_regressor = xgb, use_features_in_secondary = True)","metadata":{"execution":{"iopub.status.busy":"2022-07-07T13:42:24.441911Z","iopub.execute_input":"2022-07-07T13:42:24.442513Z","iopub.status.idle":"2022-07-07T13:42:24.448360Z","shell.execute_reply.started":"2022-07-07T13:42:24.442475Z","shell.execute_reply":"2022-07-07T13:42:24.447418Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"stacked_model = stacked.fit(np.array(x_train), np.array(y_train))\n\n#RMSLE score of the stacked model on full train data\nstacked_score = rmsle(y_train, stacked_model.predict(x_train))\nprint(\"RMSLE score of stacked models on full data:\", stacked_score)","metadata":{"execution":{"iopub.status.busy":"2022-07-07T13:42:24.449694Z","iopub.execute_input":"2022-07-07T13:42:24.450376Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred = np.floor(np.expm1(stacked_model.predict(x_test)))\ny_pred[0:5]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = pd.DataFrame()\nsubmission[\"Id\"] = test_id\nsubmission[\"SalePrice\"] = y_pred\nsubmission.to_csv(\"submission.csv\", index = False)\nsubmission.head(n = 10)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}