{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-11T20:15:40.289143Z","iopub.execute_input":"2022-07-11T20:15:40.290250Z","iopub.status.idle":"2022-07-11T20:15:40.325865Z","shell.execute_reply.started":"2022-07-11T20:15:40.290119Z","shell.execute_reply":"2022-07-11T20:15:40.324636Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import scipy.stats\nimport pickle\nimport matplotlib.pyplot as plt","metadata":{"execution":{"iopub.status.busy":"2022-07-11T20:18:21.097453Z","iopub.execute_input":"2022-07-11T20:18:21.098140Z","iopub.status.idle":"2022-07-11T20:18:21.103551Z","shell.execute_reply.started":"2022-07-11T20:18:21.098093Z","shell.execute_reply":"2022-07-11T20:18:21.102621Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nfrom sklearn.model_selection import KFold\nfrom sklearn.model_selection import LeaveOneOut\nfrom sklearn.model_selection import cross_val_score\nfrom sklearn.model_selection import cross_validate\n\nfrom sklearn.compose import ColumnTransformer\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.impute import SimpleImputer\n\n# Preprocessing\nfrom sklearn.preprocessing import OneHotEncoder, LabelEncoder, OrdinalEncoder\nfrom sklearn.preprocessing import MinMaxScaler\nfrom sklearn.preprocessing import StandardScaler\n\n# dimensionality reduction\nfrom sklearn.decomposition import PCA\n\n# Hyperparameters Search \nfrom sklearn.model_selection import RandomizedSearchCV\n\n\n## Feature Selection RFE\nfrom sklearn.feature_selection import RFE\nfrom sklearn.feature_selection import RFECV\n\n## Regression\nfrom sklearn.linear_model import LinearRegression\nfrom sklearn.linear_model import Lasso\nfrom sklearn.linear_model import LassoCV, RidgeCV\nfrom sklearn.ensemble import RandomForestRegressor\nfrom sklearn.ensemble import StackingRegressor\nfrom xgboost import XGBRegressor\n#Decision Tree regressor\nfrom sklearn.tree import DecisionTreeRegressor\nfrom sklearn.svm import LinearSVR\n## Isolation Forest\nfrom sklearn.ensemble import IsolationForest\n\n# metrics\nfrom sklearn.metrics import mean_squared_error, r2_score","metadata":{"execution":{"iopub.status.busy":"2022-07-11T20:15:41.128693Z","iopub.execute_input":"2022-07-11T20:15:41.129046Z","iopub.status.idle":"2022-07-11T20:15:41.897465Z","shell.execute_reply.started":"2022-07-11T20:15:41.129015Z","shell.execute_reply":"2022-07-11T20:15:41.895926Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Importing Data","metadata":{}},{"cell_type":"code","source":"df_train = pd.read_csv(\"/kaggle/input/house-prices-advanced-regression-techniques/train.csv\")\ndf_test = pd.read_csv(\"/kaggle/input/house-prices-advanced-regression-techniques/test.csv\")\ndf_sample_sub = pd.read_csv(\"/kaggle/input/house-prices-advanced-regression-techniques/sample_submission.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-07-11T20:15:41.902954Z","iopub.execute_input":"2022-07-11T20:15:41.906418Z","iopub.status.idle":"2022-07-11T20:15:42.014946Z","shell.execute_reply.started":"2022-07-11T20:15:41.906367Z","shell.execute_reply":"2022-07-11T20:15:42.013656Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Viewing headers/columns","metadata":{}},{"cell_type":"code","source":"df_train.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-11T20:15:42.016908Z","iopub.execute_input":"2022-07-11T20:15:42.018693Z","iopub.status.idle":"2022-07-11T20:15:42.066602Z","shell.execute_reply.started":"2022-07-11T20:15:42.018640Z","shell.execute_reply":"2022-07-11T20:15:42.064306Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.columns","metadata":{"execution":{"iopub.status.busy":"2022-07-11T20:15:42.068707Z","iopub.execute_input":"2022-07-11T20:15:42.069487Z","iopub.status.idle":"2022-07-11T20:15:42.080971Z","shell.execute_reply.started":"2022-07-11T20:15:42.069438Z","shell.execute_reply":"2022-07-11T20:15:42.079249Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_sample_sub.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-11T20:15:42.082712Z","iopub.execute_input":"2022-07-11T20:15:42.085013Z","iopub.status.idle":"2022-07-11T20:15:42.101957Z","shell.execute_reply.started":"2022-07-11T20:15:42.084966Z","shell.execute_reply":"2022-07-11T20:15:42.100563Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-11T20:15:42.104351Z","iopub.execute_input":"2022-07-11T20:15:42.105176Z","iopub.status.idle":"2022-07-11T20:15:42.157035Z","shell.execute_reply.started":"2022-07-11T20:15:42.105134Z","shell.execute_reply":"2022-07-11T20:15:42.155236Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-11T20:15:42.158630Z","iopub.execute_input":"2022-07-11T20:15:42.158976Z","iopub.status.idle":"2022-07-11T20:15:42.198847Z","shell.execute_reply.started":"2022-07-11T20:15:42.158947Z","shell.execute_reply":"2022-07-11T20:15:42.197552Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Dropping columns with too many nan values","metadata":{}},{"cell_type":"code","source":"nan_columns_train = df_train.isna().sum().sort_values(ascending=False)\nnan_columns_test = df_test.isna().sum().sort_values(ascending=False)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-11T20:15:42.204157Z","iopub.execute_input":"2022-07-11T20:15:42.206326Z","iopub.status.idle":"2022-07-11T20:15:42.276015Z","shell.execute_reply.started":"2022-07-11T20:15:42.206269Z","shell.execute_reply":"2022-07-11T20:15:42.274369Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"too_many_nans = nan_columns_test[:4].index.tolist()\ntoo_many_nans ","metadata":{"execution":{"iopub.status.busy":"2022-07-11T20:15:42.278900Z","iopub.execute_input":"2022-07-11T20:15:42.280823Z","iopub.status.idle":"2022-07-11T20:15:42.291921Z","shell.execute_reply.started":"2022-07-11T20:15:42.280769Z","shell.execute_reply":"2022-07-11T20:15:42.290870Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.drop(columns=too_many_nans, inplace=True)\ndf_test.drop(columns=too_many_nans, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-11T20:15:42.293267Z","iopub.execute_input":"2022-07-11T20:15:42.294074Z","iopub.status.idle":"2022-07-11T20:15:42.306883Z","shell.execute_reply.started":"2022-07-11T20:15:42.294043Z","shell.execute_reply":"2022-07-11T20:15:42.305661Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Getting Correlations","metadata":{}},{"cell_type":"code","source":"df_train.corr()","metadata":{"execution":{"iopub.status.busy":"2022-07-11T20:15:42.308433Z","iopub.execute_input":"2022-07-11T20:15:42.309271Z","iopub.status.idle":"2022-07-11T20:15:42.373358Z","shell.execute_reply.started":"2022-07-11T20:15:42.309229Z","shell.execute_reply":"2022-07-11T20:15:42.372424Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"correlation_price = np.abs(df_train.corr()[\"SalePrice\"]).sort_values()\ncorrelation_price ","metadata":{"execution":{"iopub.status.busy":"2022-07-11T20:15:42.374864Z","iopub.execute_input":"2022-07-11T20:15:42.375480Z","iopub.status.idle":"2022-07-11T20:15:42.393070Z","shell.execute_reply.started":"2022-07-11T20:15:42.375448Z","shell.execute_reply":"2022-07-11T20:15:42.392267Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Dropping Features that are not correlated to Price","metadata":{}},{"cell_type":"code","source":"columns_to_drop = list(correlation_price[correlation_price < 0.022].index)\ncolumns_to_drop ","metadata":{"execution":{"iopub.status.busy":"2022-07-11T20:16:54.198488Z","iopub.execute_input":"2022-07-11T20:16:54.199105Z","iopub.status.idle":"2022-07-11T20:16:54.210991Z","shell.execute_reply.started":"2022-07-11T20:16:54.199053Z","shell.execute_reply":"2022-07-11T20:16:54.209317Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.drop(columns=columns_to_drop, inplace=True)\ndf_test.drop(columns=columns_to_drop, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-11T20:16:54.464360Z","iopub.execute_input":"2022-07-11T20:16:54.465900Z","iopub.status.idle":"2022-07-11T20:16:54.479276Z","shell.execute_reply.started":"2022-07-11T20:16:54.465842Z","shell.execute_reply":"2022-07-11T20:16:54.477580Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Viewing Highly-correlated pairs","metadata":{}},{"cell_type":"code","source":"correlation = df_train.corr().abs()\n\nunstack_corr = correlation.unstack()\ncorrelated_pairs = unstack_corr.sort_values(kind=\"quicksort\", ascending=False)\nhigh_corr_pairs = correlated_pairs[(correlated_pairs < 1.0) & (correlated_pairs  >= 0.8)] \nhigh_corr_pairs","metadata":{"execution":{"iopub.status.busy":"2022-07-11T20:16:55.603142Z","iopub.execute_input":"2022-07-11T20:16:55.603945Z","iopub.status.idle":"2022-07-11T20:16:55.623531Z","shell.execute_reply.started":"2022-07-11T20:16:55.603898Z","shell.execute_reply":"2022-07-11T20:16:55.622372Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for col_name in high_corr_pairs.index.get_level_values(0):\n    print(col_name)\n    print(df_train[col_name].isna().sum())\n    print(df_test[col_name].isna().sum())","metadata":{"execution":{"iopub.status.busy":"2022-07-11T20:16:55.944104Z","iopub.execute_input":"2022-07-11T20:16:55.944819Z","iopub.status.idle":"2022-07-11T20:16:55.957732Z","shell.execute_reply.started":"2022-07-11T20:16:55.944771Z","shell.execute_reply":"2022-07-11T20:16:55.956206Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Replacing nan values with the corresponding values of the highly correlated feature","metadata":{}},{"cell_type":"code","source":"null_garage_train = df_train.GarageYrBlt.isna()\nnull_garage_test = df_train.GarageYrBlt.isna()\nnull_garate_ind_train = null_garage_train[null_garage_train==True].index\nnull_garate_ind_test = null_garage_test[null_garage_test==True].index","metadata":{"execution":{"iopub.status.busy":"2022-07-11T20:16:57.405066Z","iopub.execute_input":"2022-07-11T20:16:57.405672Z","iopub.status.idle":"2022-07-11T20:16:57.416672Z","shell.execute_reply.started":"2022-07-11T20:16:57.405623Z","shell.execute_reply":"2022-07-11T20:16:57.415627Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.GarageYrBlt.iloc[null_garate_ind_train] = df_train.YearBuilt.iloc[null_garate_ind_train]","metadata":{"execution":{"iopub.status.busy":"2022-07-11T20:16:57.708369Z","iopub.execute_input":"2022-07-11T20:16:57.709120Z","iopub.status.idle":"2022-07-11T20:16:57.718969Z","shell.execute_reply.started":"2022-07-11T20:16:57.709074Z","shell.execute_reply":"2022-07-11T20:16:57.717370Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test.GarageYrBlt.iloc[null_garate_ind_test] = df_test.YearBuilt.iloc[null_garate_ind_test]","metadata":{"execution":{"iopub.status.busy":"2022-07-11T20:16:58.122323Z","iopub.execute_input":"2022-07-11T20:16:58.122772Z","iopub.status.idle":"2022-07-11T20:16:58.130627Z","shell.execute_reply.started":"2022-07-11T20:16:58.122742Z","shell.execute_reply":"2022-07-11T20:16:58.129528Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Checking the distribution of the Target column SalePrice","metadata":{}},{"cell_type":"code","source":"ax = plt.figure(figsize=(20,20))\ndf_train.hist(ax=ax);","metadata":{"execution":{"iopub.status.busy":"2022-07-11T20:19:07.122689Z","iopub.execute_input":"2022-07-11T20:19:07.123141Z","iopub.status.idle":"2022-07-11T20:19:11.768370Z","shell.execute_reply.started":"2022-07-11T20:19:07.123107Z","shell.execute_reply":"2022-07-11T20:19:11.766813Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"num_train =  df_train.select_dtypes(include=[\"number\"]).dropna(axis=0)\ntrain_data = num_train.iloc[:,:-1]\ntarget = num_train.iloc[:,-1]\n","metadata":{"execution":{"iopub.status.busy":"2022-07-11T20:25:03.937248Z","iopub.execute_input":"2022-07-11T20:25:03.937670Z","iopub.status.idle":"2022-07-11T20:25:03.953058Z","shell.execute_reply.started":"2022-07-11T20:25:03.937637Z","shell.execute_reply":"2022-07-11T20:25:03.951320Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"reg = LinearRegression()\nscaler = StandardScaler()\n\npipeline = Pipeline([(\"scaler\", scaler), (\"reg\", reg)])\npipeline.fit(train_data, target) ","metadata":{"execution":{"iopub.status.busy":"2022-07-11T20:25:15.759048Z","iopub.execute_input":"2022-07-11T20:25:15.759480Z","iopub.status.idle":"2022-07-11T20:25:15.806383Z","shell.execute_reply.started":"2022-07-11T20:25:15.759451Z","shell.execute_reply":"2022-07-11T20:25:15.805031Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"prediction = pipeline.predict(train_data)\nplt.hist(prediction)","metadata":{"execution":{"iopub.status.busy":"2022-07-11T20:29:31.288108Z","iopub.execute_input":"2022-07-11T20:29:31.288625Z","iopub.status.idle":"2022-07-11T20:29:31.529127Z","shell.execute_reply.started":"2022-07-11T20:29:31.288586Z","shell.execute_reply":"2022-07-11T20:29:31.527973Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"scipy.stats.kurtosis(prediction), scipy.stats.skew(prediction)","metadata":{"execution":{"iopub.status.busy":"2022-07-11T20:29:32.371247Z","iopub.execute_input":"2022-07-11T20:29:32.371697Z","iopub.status.idle":"2022-07-11T20:29:32.380582Z","shell.execute_reply.started":"2022-07-11T20:29:32.371664Z","shell.execute_reply":"2022-07-11T20:29:32.379055Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"target.hist();","metadata":{"execution":{"iopub.status.busy":"2022-07-11T20:30:11.136054Z","iopub.execute_input":"2022-07-11T20:30:11.136551Z","iopub.status.idle":"2022-07-11T20:30:11.356755Z","shell.execute_reply.started":"2022-07-11T20:30:11.136514Z","shell.execute_reply":"2022-07-11T20:30:11.355447Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"scipy.stats.kurtosis(target), scipy.stats.skew(target)","metadata":{"execution":{"iopub.status.busy":"2022-07-11T20:29:51.063748Z","iopub.execute_input":"2022-07-11T20:29:51.064192Z","iopub.status.idle":"2022-07-11T20:29:51.074428Z","shell.execute_reply.started":"2022-07-11T20:29:51.064160Z","shell.execute_reply":"2022-07-11T20:29:51.073169Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Caution, the target distribution is very skewed. \n\nNote that the prediction distribution looks more normal that the target disribution, hence for better prediction it may be better to transform it to normal.\n\nRecall that in the linear regression model we have\n\n$$ y = \\beta_0 + \\sum_{i=1}^n \\beta_i X_i + \\varepsilon,$$ \n\nwhere $\\varepsilon \\sim \\mathcal{N}(0, \\sigma^2)$ for some $\\sigma>0$ given the values $X_i = x_i$.\n\nHence, the only underlying normality is about the residual error. \n\nIf $X_i$ were independent and identically distributed (iid) then for large n, the sum $\\sum_{i=1}^n \\beta_i X_i$ will be approximately normal. \n\n\n","metadata":{}},{"cell_type":"code","source":"np.log(df_train.SalePrice).hist(); #looks much more normal","metadata":{"execution":{"iopub.status.busy":"2022-07-11T20:42:54.597921Z","iopub.execute_input":"2022-07-11T20:42:54.598431Z","iopub.status.idle":"2022-07-11T20:42:54.804140Z","shell.execute_reply.started":"2022-07-11T20:42:54.598394Z","shell.execute_reply":"2022-07-11T20:42:54.802580Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"scipy.stats.kurtosis(np.log(df_train.SalePrice)), scipy.stats.skew(np.log(df_train.SalePrice))","metadata":{"execution":{"iopub.status.busy":"2022-07-11T20:42:54.840953Z","iopub.execute_input":"2022-07-11T20:42:54.841798Z","iopub.status.idle":"2022-07-11T20:42:54.852815Z","shell.execute_reply.started":"2022-07-11T20:42:54.841758Z","shell.execute_reply":"2022-07-11T20:42:54.851208Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"box_cox_transformed = scipy.stats.boxcox(df_train.SalePrice)[0]","metadata":{"execution":{"iopub.status.busy":"2022-07-11T20:53:43.944716Z","iopub.execute_input":"2022-07-11T20:53:43.945261Z","iopub.status.idle":"2022-07-11T20:53:43.954091Z","shell.execute_reply.started":"2022-07-11T20:53:43.945224Z","shell.execute_reply":"2022-07-11T20:53:43.953306Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.hist(box_cox_transformed);","metadata":{"execution":{"iopub.status.busy":"2022-07-11T20:55:23.510494Z","iopub.execute_input":"2022-07-11T20:55:23.510975Z","iopub.status.idle":"2022-07-11T20:55:24.526637Z","shell.execute_reply.started":"2022-07-11T20:55:23.510940Z","shell.execute_reply":"2022-07-11T20:55:24.525626Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"scipy.stats.kurtosis(box_cox_transformed), scipy.stats.skew(box_cox_transformed)","metadata":{"execution":{"iopub.status.busy":"2022-07-11T20:53:47.430350Z","iopub.execute_input":"2022-07-11T20:53:47.431028Z","iopub.status.idle":"2022-07-11T20:53:47.439992Z","shell.execute_reply.started":"2022-07-11T20:53:47.430990Z","shell.execute_reply":"2022-07-11T20:53:47.438639Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = df_train.iloc[:, :-1]\ny = np.log(df_train.SalePrice) # transforming the target, we will need to reverse this operation later on","metadata":{"execution":{"iopub.status.busy":"2022-07-11T20:55:29.909427Z","iopub.execute_input":"2022-07-11T20:55:29.910002Z","iopub.status.idle":"2022-07-11T20:55:29.920363Z","shell.execute_reply.started":"2022-07-11T20:55:29.909954Z","shell.execute_reply":"2022-07-11T20:55:29.919131Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Preprocessing for the pipeline","metadata":{}},{"cell_type":"code","source":"num_columns =  X.select_dtypes(include=[\"number\"]).columns # selecting numerical columns\nnon_num_columns = X.select_dtypes(exclude=[\"number\"]).columns # selecting non-num columns","metadata":{"execution":{"iopub.status.busy":"2022-07-11T20:55:30.443432Z","iopub.execute_input":"2022-07-11T20:55:30.444901Z","iopub.status.idle":"2022-07-11T20:55:30.459268Z","shell.execute_reply.started":"2022-07-11T20:55:30.444847Z","shell.execute_reply":"2022-07-11T20:55:30.457822Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X.columns.shape, len(num_columns) + len(non_num_columns)","metadata":{"execution":{"iopub.status.busy":"2022-07-11T20:55:30.686822Z","iopub.execute_input":"2022-07-11T20:55:30.687484Z","iopub.status.idle":"2022-07-11T20:55:30.695758Z","shell.execute_reply.started":"2022-07-11T20:55:30.687449Z","shell.execute_reply":"2022-07-11T20:55:30.694656Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train, X_val, y_train, y_val = train_test_split(X, y, test_size=0.15, random_state=0)","metadata":{"execution":{"iopub.status.busy":"2022-07-11T20:55:30.922716Z","iopub.execute_input":"2022-07-11T20:55:30.923544Z","iopub.status.idle":"2022-07-11T20:55:30.931985Z","shell.execute_reply.started":"2022-07-11T20:55:30.923499Z","shell.execute_reply":"2022-07-11T20:55:30.931130Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"numerical_transformer = SimpleImputer(strategy='median') # in numerical we replace nans with median\n\ncategorical_transformer = Pipeline(\n    steps=[\n        (\"encoder\", OrdinalEncoder(handle_unknown='use_encoded_value', unknown_value=-1)), \n        (\"imputer\", SimpleImputer(strategy=\"most_frequent\"))]) # with numerical we replace the nan with mode\n\npreprocessor = ColumnTransformer(\n    transformers=[\n        ('num', numerical_transformer, num_columns),\n        ('cat', categorical_transformer, non_num_columns)\n    ])","metadata":{"execution":{"iopub.status.busy":"2022-07-11T20:55:31.107472Z","iopub.execute_input":"2022-07-11T20:55:31.108061Z","iopub.status.idle":"2022-07-11T20:55:31.114137Z","shell.execute_reply.started":"2022-07-11T20:55:31.108029Z","shell.execute_reply":"2022-07-11T20:55:31.113311Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Linear Regression with PCA","metadata":{}},{"cell_type":"code","source":"def reg_pca(k):\n    pca = PCA(n_components=k)\n    reg = LinearRegression()\n    scaler = StandardScaler(with_std=False)\n    pipeline = Pipeline([('preprocess', preprocessor), (\"scaler\", scaler), (\"pca\", pca), (\"reg\", reg)])\n    pipeline.fit(X_train, y_train) \n    y_pred = pipeline.predict(X_val)\n    rmsescore =  mean_squared_error(y_val, y_pred, squared=False)\n    return rmsescore","metadata":{"execution":{"iopub.status.busy":"2022-07-11T20:55:31.547412Z","iopub.execute_input":"2022-07-11T20:55:31.548022Z","iopub.status.idle":"2022-07-11T20:55:31.555064Z","shell.execute_reply.started":"2022-07-11T20:55:31.547990Z","shell.execute_reply":"2022-07-11T20:55:31.554144Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"best_k_score = np.inf\nbest_k = 1\nfor k in range(1, 60):\n    if reg_pca(k) < best_k_score:\n        best_k_score = reg_pca(k)\n        best_k = k","metadata":{"execution":{"iopub.status.busy":"2022-07-11T20:55:32.021079Z","iopub.execute_input":"2022-07-11T20:55:32.021700Z","iopub.status.idle":"2022-07-11T20:55:51.254051Z","shell.execute_reply.started":"2022-07-11T20:55:32.021667Z","shell.execute_reply":"2022-07-11T20:55:51.252587Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(best_k_score, best_k)","metadata":{"execution":{"iopub.status.busy":"2022-07-11T20:55:51.256765Z","iopub.execute_input":"2022-07-11T20:55:51.259531Z","iopub.status.idle":"2022-07-11T20:55:51.267700Z","shell.execute_reply.started":"2022-07-11T20:55:51.259475Z","shell.execute_reply":"2022-07-11T20:55:51.266536Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Lasso Regresssion (that performs feature selection)","metadata":{}},{"cell_type":"code","source":"my_alphas = np.arange(0.001, 0.01, 0.001) # start, end + step, step\nscaler = StandardScaler()\nreg = LassoCV(alphas=my_alphas, cv=5)\npipeline2 = Pipeline([('preprocess', preprocessor), ('scaler', scaler), ('reg', reg)])\npipeline2.fit(X_train, y_train) \n\nbest_alpha = pipeline2[\"reg\"].alpha_\nprint(f\"best alpha is {best_alpha}\")","metadata":{"execution":{"iopub.status.busy":"2022-07-11T20:55:51.269430Z","iopub.execute_input":"2022-07-11T20:55:51.271800Z","iopub.status.idle":"2022-07-11T20:55:51.447739Z","shell.execute_reply.started":"2022-07-11T20:55:51.271751Z","shell.execute_reply":"2022-07-11T20:55:51.446065Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred = pipeline2.predict(X_train)\nrmsescore_train =  mean_squared_error(y_train, y_pred, squared=False)\nprint(f\"R2 score on training data set is {pipeline2.score(X_train, y_train)}\")\nprint(f\"MSE score on training data set is {rmsescore_train }\")\n\n# making predictions on the test data set and printing relevant scores\ny_pred = pipeline2.predict(X_val)\nr2score =  r2_score(y_val, y_pred)\nrmsescore =  mean_squared_error(y_val, y_pred, squared=False)\nprint(f\"R2 score on test data set is {r2score}\")\nprint(f\"RMSE score on test data set is {rmsescore}\")","metadata":{"execution":{"iopub.status.busy":"2022-07-11T20:55:51.451076Z","iopub.execute_input":"2022-07-11T20:55:51.454181Z","iopub.status.idle":"2022-07-11T20:55:51.645560Z","shell.execute_reply.started":"2022-07-11T20:55:51.454120Z","shell.execute_reply":"2022-07-11T20:55:51.644245Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"lasso_coeffs = pipeline2[\"reg\"].coef_\nlasso_coeffs ","metadata":{"execution":{"iopub.status.busy":"2022-07-11T20:55:51.647446Z","iopub.execute_input":"2022-07-11T20:55:51.648235Z","iopub.status.idle":"2022-07-11T20:55:51.658893Z","shell.execute_reply.started":"2022-07-11T20:55:51.648169Z","shell.execute_reply":"2022-07-11T20:55:51.657606Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"good_coeffs = np.where(lasso_coeffs != 0.)[0] # these are the best features in the model\ngood_coeffs, good_coeffs.shape[0]","metadata":{"execution":{"iopub.status.busy":"2022-07-11T20:55:51.661114Z","iopub.execute_input":"2022-07-11T20:55:51.661938Z","iopub.status.idle":"2022-07-11T20:55:51.672643Z","shell.execute_reply.started":"2022-07-11T20:55:51.661888Z","shell.execute_reply":"2022-07-11T20:55:51.671335Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Ridge regression model (L2 regularization)","metadata":{}},{"cell_type":"code","source":"my_alphas = np.arange(0.001, 0.01, 0.001) # start, end + step, step\nscaler = StandardScaler()\nreg = RidgeCV(alphas=my_alphas, cv=5)\npipeline3 = Pipeline([('preprocess', preprocessor), ('scaler', scaler), ('reg', reg)])\npipeline3.fit(X_train, y_train) \n\nbest_alpha = pipeline2[\"reg\"].alpha_\nprint(f\"best alpha is {best_alpha}\")","metadata":{"execution":{"iopub.status.busy":"2022-07-11T20:55:53.048677Z","iopub.execute_input":"2022-07-11T20:55:53.049281Z","iopub.status.idle":"2022-07-11T20:55:53.408998Z","shell.execute_reply.started":"2022-07-11T20:55:53.049248Z","shell.execute_reply":"2022-07-11T20:55:53.407706Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred = pipeline3.predict(X_train)\nrmsescore_train =  mean_squared_error(y_train, y_pred, squared=False)\nprint(f\"R2 score on training data set is {pipeline3.score(X_train, y_train)}\")\nprint(f\"MSE score on training data set is {rmsescore_train }\")\n\n# making predictions on the test data set and printing relevant scores\ny_pred = pipeline3.predict(X_val)\nr2score =  r2_score(y_val, y_pred)\nrmsescore =  mean_squared_error(y_val, y_pred, squared=False)\nprint(f\"R2 score on test data set is {r2score}\")\nprint(f\"RMSE score on test data set is {rmsescore}\")","metadata":{"execution":{"iopub.status.busy":"2022-07-11T20:55:53.506838Z","iopub.execute_input":"2022-07-11T20:55:53.507395Z","iopub.status.idle":"2022-07-11T20:55:53.638730Z","shell.execute_reply.started":"2022-07-11T20:55:53.507348Z","shell.execute_reply":"2022-07-11T20:55:53.637442Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Model that stacks regressor together (uncomment if you wish to run, it is slow)","metadata":{}},{"cell_type":"code","source":"# estimators = [('lr', LinearRegression()), ('rf', RandomForestRegressor(random_state=0)), ('xgb', XGBRegressor())]\n# reg = StackingRegressor(\n#     estimators=estimators,\n#     final_estimator=RidgeCV()\n# )\n# pca = PCA(n_components=best_k)\n# scaler = StandardScaler()\n# pipeline = Pipeline([('preprocess', preprocessor), (\"scaler\", scaler), (\"reg\", reg)])\n# #Do the regression (fit the parameter of the model)\n# pipeline.fit(X_train, y_train) \n","metadata":{"execution":{"iopub.status.busy":"2022-07-11T20:54:25.854411Z","iopub.execute_input":"2022-07-11T20:54:25.855174Z","iopub.status.idle":"2022-07-11T20:54:25.862557Z","shell.execute_reply.started":"2022-07-11T20:54:25.855125Z","shell.execute_reply":"2022-07-11T20:54:25.861063Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# y_pred = pipeline.predict(X_train)\n# rmsescore_train =  mean_squared_error(y_train, y_pred, squared=False)\n# print(f\"R2 score on training data set is {pipeline.score(X_train, y_train)}\")\n# print(f\"MSE score on training data set is {rmsescore_train }\")\n\n# # making predictions on the test data set and printing relevant scores\n# y_pred = pipeline.predict(X_val)\n# r2score =  r2_score(y_val, y_pred)\n# rmsescore =  mean_squared_error(y_val, y_pred, squared=False)\n# print(f\"R2 score on test data set is {r2score}\")\n# print(f\"RMSE score on test data set is {rmsescore}\")","metadata":{"execution":{"iopub.status.busy":"2022-07-11T20:54:25.865522Z","iopub.execute_input":"2022-07-11T20:54:25.866287Z","iopub.status.idle":"2022-07-11T20:54:25.874581Z","shell.execute_reply.started":"2022-07-11T20:54:25.866238Z","shell.execute_reply":"2022-07-11T20:54:25.873391Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Final Pipeline for submission ","metadata":{}},{"cell_type":"code","source":"reg = Lasso(alpha=best_alpha)\nscaler = StandardScaler()\npipeline1 = Pipeline([('preprocess', preprocessor), (\"scaler\", scaler), (\"reg\", reg)])\npipeline1.fit(X_train, y_train)#, clf__sample_weight=classes_weights)\ny_pred = pipeline1.predict(X_val)\nmean_squared_error(y_val, y_pred, squared=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-11T20:54:26.787745Z","iopub.execute_input":"2022-07-11T20:54:26.788344Z","iopub.status.idle":"2022-07-11T20:54:26.909476Z","shell.execute_reply.started":"2022-07-11T20:54:26.788298Z","shell.execute_reply":"2022-07-11T20:54:26.907797Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# saving the model via pickle\nfilename = '/kaggle/working/finalized_model.sav'\npickle.dump(pipeline1, open(filename, 'wb'))","metadata":{"execution":{"iopub.status.busy":"2022-07-11T20:54:36.691503Z","iopub.execute_input":"2022-07-11T20:54:36.691945Z","iopub.status.idle":"2022-07-11T20:54:36.701070Z","shell.execute_reply.started":"2022-07-11T20:54:36.691910Z","shell.execute_reply":"2022-07-11T20:54:36.699575Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"loaded_model = pickle.load(open(filename, 'rb'))","metadata":{"execution":{"iopub.status.busy":"2022-07-11T20:54:37.006898Z","iopub.execute_input":"2022-07-11T20:54:37.007747Z","iopub.status.idle":"2022-07-11T20:54:37.016459Z","shell.execute_reply.started":"2022-07-11T20:54:37.007709Z","shell.execute_reply":"2022-07-11T20:54:37.013613Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"loaded_model.fit(X_train, y_train)#, clf__sample_weight=classes_weights) ","metadata":{"execution":{"iopub.status.busy":"2022-07-11T20:54:37.530141Z","iopub.execute_input":"2022-07-11T20:54:37.531492Z","iopub.status.idle":"2022-07-11T20:54:37.658909Z","shell.execute_reply.started":"2022-07-11T20:54:37.531419Z","shell.execute_reply":"2022-07-11T20:54:37.657673Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_submission = pd.DataFrame()\ndf_submission[\"Id\"] = df_sample_sub.Id","metadata":{"execution":{"iopub.status.busy":"2022-07-11T20:54:46.841237Z","iopub.execute_input":"2022-07-11T20:54:46.841670Z","iopub.status.idle":"2022-07-11T20:54:46.850091Z","shell.execute_reply.started":"2022-07-11T20:54:46.841638Z","shell.execute_reply":"2022-07-11T20:54:46.849116Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred = loaded_model.predict(df_test)\ndf_submission[\"SalePrice\"] = np.exp(y_pred)# we need to bring the submission to the correct format","metadata":{"execution":{"iopub.status.busy":"2022-07-11T20:54:53.853647Z","iopub.execute_input":"2022-07-11T20:54:53.854104Z","iopub.status.idle":"2022-07-11T20:54:53.910802Z","shell.execute_reply.started":"2022-07-11T20:54:53.854068Z","shell.execute_reply":"2022-07-11T20:54:53.909325Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_submission.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-11T20:54:55.394648Z","iopub.execute_input":"2022-07-11T20:54:55.395128Z","iopub.status.idle":"2022-07-11T20:54:55.408183Z","shell.execute_reply.started":"2022-07-11T20:54:55.395094Z","shell.execute_reply":"2022-07-11T20:54:55.406682Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_submission.to_csv(\"/kaggle/working/submission.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-11T14:41:25.251941Z","iopub.execute_input":"2022-07-11T14:41:25.252466Z","iopub.status.idle":"2022-07-11T14:41:25.265534Z","shell.execute_reply.started":"2022-07-11T14:41:25.252435Z","shell.execute_reply":"2022-07-11T14:41:25.264172Z"},"trusted":true},"execution_count":null,"outputs":[]}]}