{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-11T05:53:01.316945Z","iopub.execute_input":"2022-08-11T05:53:01.317655Z","iopub.status.idle":"2022-08-11T05:53:01.344853Z","shell.execute_reply.started":"2022-08-11T05:53:01.317556Z","shell.execute_reply":"2022-08-11T05:53:01.344084Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Importing the Essential Libraries, Metrics\n","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nimport warnings\nwarnings.filterwarnings(\"ignore\")\n\nfrom sklearn.model_selection import train_test_split, cross_val_score\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.metrics import r2_score, mean_absolute_error, mean_squared_error\n\nfrom sklearn.linear_model import LinearRegression\nfrom sklearn.linear_model import Ridge\nfrom sklearn.linear_model import Lasso\nfrom sklearn.linear_model import ElasticNet\nfrom sklearn.ensemble import RandomForestRegressor\nfrom sklearn.svm import SVR\nfrom xgboost import XGBRegressor\nfrom sklearn.preprocessing import PolynomialFeatures","metadata":{"execution":{"iopub.status.busy":"2022-08-11T05:53:01.347321Z","iopub.execute_input":"2022-08-11T05:53:01.347736Z","iopub.status.idle":"2022-08-11T05:53:02.741702Z","shell.execute_reply.started":"2022-08-11T05:53:01.347695Z","shell.execute_reply":"2022-08-11T05:53:02.740880Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Importing dataset","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv(\"../input/house-prices-advanced-regression-techniques/train.csv\")\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-11T05:53:02.743286Z","iopub.execute_input":"2022-08-11T05:53:02.743787Z","iopub.status.idle":"2022-08-11T05:53:02.808703Z","shell.execute_reply.started":"2022-08-11T05:53:02.743745Z","shell.execute_reply":"2022-08-11T05:53:02.807945Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Exploratory Data Analysis\n","metadata":{}},{"cell_type":"markdown","source":"### Checking the shape—i.e. size—of the data\n\n","metadata":{}},{"cell_type":"code","source":"df.shape","metadata":{"execution":{"iopub.status.busy":"2022-08-11T05:53:02.809922Z","iopub.execute_input":"2022-08-11T05:53:02.810419Z","iopub.status.idle":"2022-08-11T05:53:02.819165Z","shell.execute_reply.started":"2022-08-11T05:53:02.810234Z","shell.execute_reply":"2022-08-11T05:53:02.818228Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Learning the dtypes of columns' and how many non-null values are there in those columns\n\n","metadata":{}},{"cell_type":"code","source":"df.info()","metadata":{"execution":{"iopub.status.busy":"2022-08-11T05:53:02.822368Z","iopub.execute_input":"2022-08-11T05:53:02.822709Z","iopub.status.idle":"2022-08-11T05:53:02.864440Z","shell.execute_reply.started":"2022-08-11T05:53:02.822678Z","shell.execute_reply":"2022-08-11T05:53:02.863052Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Getting the statistical summary of dataset\n\n","metadata":{}},{"cell_type":"code","source":"df.describe()\n","metadata":{"execution":{"iopub.status.busy":"2022-08-11T05:53:02.866122Z","iopub.execute_input":"2022-08-11T05:53:02.866923Z","iopub.status.idle":"2022-08-11T05:53:02.988091Z","shell.execute_reply.started":"2022-08-11T05:53:02.866875Z","shell.execute_reply":"2022-08-11T05:53:02.987012Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.describe().T # Transpose only","metadata":{"execution":{"iopub.status.busy":"2022-08-11T05:53:02.989542Z","iopub.execute_input":"2022-08-11T05:53:02.989863Z","iopub.status.idle":"2022-08-11T05:53:03.106565Z","shell.execute_reply.started":"2022-08-11T05:53:02.989834Z","shell.execute_reply":"2022-08-11T05:53:03.105460Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"(df.describe().T).shape     #total numerical features \n","metadata":{"execution":{"iopub.status.busy":"2022-08-11T05:53:03.108264Z","iopub.execute_input":"2022-08-11T05:53:03.108936Z","iopub.status.idle":"2022-08-11T05:53:03.192027Z","shell.execute_reply.started":"2022-08-11T05:53:03.108896Z","shell.execute_reply":"2022-08-11T05:53:03.191327Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Visualizing the correlations between numerical variables","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize = (10,8))\nsns.heatmap(df.corr(),cmap = 'RdBu')\nplt.title('Correlations between Variables',size = 15)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-11T05:53:03.193093Z","iopub.execute_input":"2022-08-11T05:53:03.193510Z","iopub.status.idle":"2022-08-11T05:53:04.215618Z","shell.execute_reply.started":"2022-08-11T05:53:03.193484Z","shell.execute_reply":"2022-08-11T05:53:04.214596Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Feature Selection\n","metadata":{}},{"cell_type":"markdown","source":"#### We are selecting numerical features which have more than 0.50 or less than -0.50 correlation rate based on Pearson Correlation Method—which is the default value of parameter \"method\" in corr() function. As for selecting categorical features, I selected the categorical values which I believe have significant effect on the target variable such as Heating and MSZoning. Our aim is to predict the price of the house. Hence we will take all corr with SalePrice ONLY.","metadata":{}},{"cell_type":"code","source":"df.corr()     # correlation of all variables with each others\n","metadata":{"execution":{"iopub.status.busy":"2022-08-11T05:53:04.216560Z","iopub.execute_input":"2022-08-11T05:53:04.216857Z","iopub.status.idle":"2022-08-11T05:53:04.273500Z","shell.execute_reply.started":"2022-08-11T05:53:04.216831Z","shell.execute_reply":"2022-08-11T05:53:04.271928Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.corr()['SalePrice']     # from the above correlation datafram choose only dependent variable column i.e SalePrice ","metadata":{"execution":{"iopub.status.busy":"2022-08-11T05:53:04.275769Z","iopub.execute_input":"2022-08-11T05:53:04.281347Z","iopub.status.idle":"2022-08-11T05:53:04.302287Z","shell.execute_reply.started":"2022-08-11T05:53:04.281293Z","shell.execute_reply":"2022-08-11T05:53:04.301045Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.corr()['SalePrice']>0.5 ","metadata":{"execution":{"iopub.status.busy":"2022-08-11T05:53:04.307458Z","iopub.execute_input":"2022-08-11T05:53:04.307894Z","iopub.status.idle":"2022-08-11T05:53:04.330545Z","shell.execute_reply.started":"2022-08-11T05:53:04.307860Z","shell.execute_reply":"2022-08-11T05:53:04.329568Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.corr()['SalePrice'] < -0.5","metadata":{"execution":{"iopub.status.busy":"2022-08-11T05:53:04.332075Z","iopub.execute_input":"2022-08-11T05:53:04.333325Z","iopub.status.idle":"2022-08-11T05:53:04.350139Z","shell.execute_reply.started":"2022-08-11T05:53:04.333283Z","shell.execute_reply":"2022-08-11T05:53:04.349046Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"list(df.corr()['SalePrice'][df.corr()['SalePrice']>0.5]) ","metadata":{"execution":{"iopub.status.busy":"2022-08-11T05:53:04.351776Z","iopub.execute_input":"2022-08-11T05:53:04.352537Z","iopub.status.idle":"2022-08-11T05:53:04.375583Z","shell.execute_reply.started":"2022-08-11T05:53:04.352500Z","shell.execute_reply":"2022-08-11T05:53:04.374532Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"list(df.corr()['SalePrice'][df.corr()['SalePrice']>0.5].index)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T05:53:04.376934Z","iopub.execute_input":"2022-08-11T05:53:04.378826Z","iopub.status.idle":"2022-08-11T05:53:04.405080Z","shell.execute_reply.started":"2022-08-11T05:53:04.378780Z","shell.execute_reply":"2022-08-11T05:53:04.404112Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(list(df.corr()['SalePrice'][(df.corr()['SalePrice']>0.5) | (df.corr()['SalePrice']<-0.5)]))","metadata":{"execution":{"iopub.status.busy":"2022-08-11T05:53:04.406711Z","iopub.execute_input":"2022-08-11T05:53:04.407753Z","iopub.status.idle":"2022-08-11T05:53:04.435065Z","shell.execute_reply.started":"2022-08-11T05:53:04.407709Z","shell.execute_reply":"2022-08-11T05:53:04.434147Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"imp_cols = list(df.corr()['SalePrice'][(df.corr()['SalePrice']>0.5) | (df.corr()['SalePrice']<-0.5)].index) # Only numerical type of features","metadata":{"execution":{"iopub.status.busy":"2022-08-11T05:53:04.436277Z","iopub.execute_input":"2022-08-11T05:53:04.437242Z","iopub.status.idle":"2022-08-11T05:53:04.499518Z","shell.execute_reply.started":"2022-08-11T05:53:04.437208Z","shell.execute_reply":"2022-08-11T05:53:04.498023Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(imp_cols)\nprint(len(imp_cols))","metadata":{"execution":{"iopub.status.busy":"2022-08-11T05:53:04.502306Z","iopub.execute_input":"2022-08-11T05:53:04.506988Z","iopub.status.idle":"2022-08-11T05:53:04.524129Z","shell.execute_reply.started":"2022-08-11T05:53:04.506915Z","shell.execute_reply":"2022-08-11T05:53:04.522122Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['KitchenQual']","metadata":{"execution":{"iopub.status.busy":"2022-08-11T05:53:04.526625Z","iopub.execute_input":"2022-08-11T05:53:04.527829Z","iopub.status.idle":"2022-08-11T05:53:04.563007Z","shell.execute_reply.started":"2022-08-11T05:53:04.527736Z","shell.execute_reply":"2022-08-11T05:53:04.561840Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cat_cols = [\"MSZoning\", \"Utilities\",\"BldgType\",\"Heating\",\"KitchenQual\",\"SaleCondition\",\"LandSlope\"] # Some more imp columns that\n                                                                                        # might effect on SalePrice of House","metadata":{"execution":{"iopub.status.busy":"2022-08-11T05:53:04.564462Z","iopub.execute_input":"2022-08-11T05:53:04.565125Z","iopub.status.idle":"2022-08-11T05:53:04.574717Z","shell.execute_reply.started":"2022-08-11T05:53:04.565082Z","shell.execute_reply":"2022-08-11T05:53:04.573848Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Redfining df to get important features for the regression analysis\nimp = imp_cols + cat_cols\nprint(imp)\nprint(len(imp))\ndf = df[imp] # df[list items from the original dataset]\ndf","metadata":{"execution":{"iopub.status.busy":"2022-08-11T05:53:04.578687Z","iopub.execute_input":"2022-08-11T05:53:04.579255Z","iopub.status.idle":"2022-08-11T05:53:04.612230Z","shell.execute_reply.started":"2022-08-11T05:53:04.579214Z","shell.execute_reply":"2022-08-11T05:53:04.610732Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Checking for the missing values\n\n","metadata":{}},{"cell_type":"code","source":"print('Missing values by column')\nprint('-'*30)\nprint(df.isna())\nprint(df.isna().sum())\nprint('-'*30)\nprint(\"Total Missing Values :\",df.isna().sum().sum())","metadata":{"execution":{"iopub.status.busy":"2022-08-11T05:53:04.614150Z","iopub.execute_input":"2022-08-11T05:53:04.614950Z","iopub.status.idle":"2022-08-11T05:53:04.642643Z","shell.execute_reply.started":"2022-08-11T05:53:04.614908Z","shell.execute_reply":"2022-08-11T05:53:04.641489Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data Visualization\n","metadata":{}},{"cell_type":"markdown","source":"### Visualizing the Correlation between the numerical variables using pairplot visualization\n\n","metadata":{}},{"cell_type":"code","source":"sns.pairplot(df[imp_cols])","metadata":{"execution":{"iopub.status.busy":"2022-08-11T05:53:04.644264Z","iopub.execute_input":"2022-08-11T05:53:04.644598Z","iopub.status.idle":"2022-08-11T05:53:04.649942Z","shell.execute_reply.started":"2022-08-11T05:53:04.644568Z","shell.execute_reply":"2022-08-11T05:53:04.648634Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Visualizing the Correlation between each column and the target variable using jointplot visualization\n\n","metadata":{}},{"cell_type":"code","source":"imp_col1 = imp_cols\nimp_col1.remove('SalePrice')\nimp_col1","metadata":{"execution":{"iopub.status.busy":"2022-08-11T05:53:04.651439Z","iopub.execute_input":"2022-08-11T05:53:04.652398Z","iopub.status.idle":"2022-08-11T05:53:04.660514Z","shell.execute_reply.started":"2022-08-11T05:53:04.652359Z","shell.execute_reply":"2022-08-11T05:53:04.659703Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# We always plot the joint plot between the dependent variable and the independent variables. \n# like here we will plot 10 jointplots btw imp_col and SalePrice\nfor i in imp_col1:\n    sns.jointplot(x=df[i],y = df['SalePrice'],kind = 'kde')","metadata":{"execution":{"iopub.status.busy":"2022-08-11T05:53:04.662130Z","iopub.execute_input":"2022-08-11T05:53:04.662940Z","iopub.status.idle":"2022-08-11T05:53:04.668434Z","shell.execute_reply.started":"2022-08-11T05:53:04.662886Z","shell.execute_reply":"2022-08-11T05:53:04.667449Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# X, y Split\n","metadata":{}},{"cell_type":"markdown","source":"### Splitting the data into X and y chunks\n\n","metadata":{}},{"cell_type":"code","source":"x = df.drop('SalePrice',axis = 1)\ny = df['SalePrice']\nprint(x)\nprint(y)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T05:53:04.669848Z","iopub.execute_input":"2022-08-11T05:53:04.670161Z","iopub.status.idle":"2022-08-11T05:53:04.689119Z","shell.execute_reply.started":"2022-08-11T05:53:04.670132Z","shell.execute_reply":"2022-08-11T05:53:04.687720Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# One-Hot Encoding\n","metadata":{}},{"cell_type":"markdown","source":"### Encoding the categorical features in X dataset by using One-Hot Encoding method\n\n","metadata":{}},{"cell_type":"markdown","source":"##### We are having some categorical variables in the df dataframe. TO convert them in to numnerical value we will use one hot encoding","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x = pd.get_dummies(x, columns = cat_cols)\nx","metadata":{"execution":{"iopub.status.busy":"2022-08-11T05:53:04.692290Z","iopub.execute_input":"2022-08-11T05:53:04.693074Z","iopub.status.idle":"2022-08-11T05:53:04.731941Z","shell.execute_reply.started":"2022-08-11T05:53:04.693040Z","shell.execute_reply":"2022-08-11T05:53:04.730902Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test = pd.read_csv(\"../input/house-prices-advanced-regression-techniques/test.csv\")\ndf_test.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-11T05:53:04.733366Z","iopub.execute_input":"2022-08-11T05:53:04.734128Z","iopub.status.idle":"2022-08-11T05:53:04.802921Z","shell.execute_reply.started":"2022-08-11T05:53:04.734095Z","shell.execute_reply":"2022-08-11T05:53:04.801855Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"imp_test = imp\nimp_test.remove('SalePrice')\nprint(imp_test)\n\ndf_test_new = df_test[imp_test]\nprint(df_test_new)\n","metadata":{"execution":{"iopub.status.busy":"2022-08-11T05:53:04.804425Z","iopub.execute_input":"2022-08-11T05:53:04.805330Z","iopub.status.idle":"2022-08-11T05:53:04.825800Z","shell.execute_reply.started":"2022-08-11T05:53:04.805287Z","shell.execute_reply":"2022-08-11T05:53:04.825024Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_test_new = df_test_new\nprint(x_test_new)\n","metadata":{"execution":{"iopub.status.busy":"2022-08-11T05:53:04.827061Z","iopub.execute_input":"2022-08-11T05:53:04.827585Z","iopub.status.idle":"2022-08-11T05:53:04.845591Z","shell.execute_reply.started":"2022-08-11T05:53:04.827555Z","shell.execute_reply":"2022-08-11T05:53:04.844426Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(cat_cols)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T05:53:04.847383Z","iopub.execute_input":"2022-08-11T05:53:04.847798Z","iopub.status.idle":"2022-08-11T05:53:04.853410Z","shell.execute_reply.started":"2022-08-11T05:53:04.847751Z","shell.execute_reply":"2022-08-11T05:53:04.852183Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_test_new = pd.get_dummies(df_test_new, columns = cat_cols)\nprint(x_test_new)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T05:53:04.854940Z","iopub.execute_input":"2022-08-11T05:53:04.855590Z","iopub.status.idle":"2022-08-11T05:53:04.892209Z","shell.execute_reply.started":"2022-08-11T05:53:04.855535Z","shell.execute_reply":"2022-08-11T05:53:04.891254Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"missing_cols = set( x.columns ) - set( x_test_new.columns )\nmissing_cols\nfor c in missing_cols:\n    x_test_new[c] = 0\nprint(x_test_new)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T05:53:04.899670Z","iopub.execute_input":"2022-08-11T05:53:04.900242Z","iopub.status.idle":"2022-08-11T05:53:04.921133Z","shell.execute_reply.started":"2022-08-11T05:53:04.900209Z","shell.execute_reply":"2022-08-11T05:53:04.920037Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Standardizing the Data\n","metadata":{}},{"cell_type":"markdown","source":"### Standardizing the numerical columns in X dataset. StandardScaler() adjusts the mean of the features as 0 and standard deviation of features as 1. ","metadata":{}},{"cell_type":"code","source":"scaler = StandardScaler()\nx[imp_col1] = scaler.fit_transform(x[imp_col1])","metadata":{"execution":{"iopub.status.busy":"2022-08-11T05:53:04.922606Z","iopub.execute_input":"2022-08-11T05:53:04.923247Z","iopub.status.idle":"2022-08-11T05:53:04.936647Z","shell.execute_reply.started":"2022-08-11T05:53:04.923188Z","shell.execute_reply":"2022-08-11T05:53:04.935583Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x","metadata":{"execution":{"iopub.status.busy":"2022-08-11T05:53:04.938066Z","iopub.execute_input":"2022-08-11T05:53:04.938798Z","iopub.status.idle":"2022-08-11T05:53:04.976184Z","shell.execute_reply.started":"2022-08-11T05:53:04.938748Z","shell.execute_reply":"2022-08-11T05:53:04.975058Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x.shape","metadata":{"execution":{"iopub.status.busy":"2022-08-11T05:53:04.977622Z","iopub.execute_input":"2022-08-11T05:53:04.977946Z","iopub.status.idle":"2022-08-11T05:53:04.983891Z","shell.execute_reply.started":"2022-08-11T05:53:04.977917Z","shell.execute_reply":"2022-08-11T05:53:04.983014Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_test_new[imp_col1] = scaler.transform(x_test_new[imp_col1])","metadata":{"execution":{"iopub.status.busy":"2022-08-11T05:53:04.985387Z","iopub.execute_input":"2022-08-11T05:53:04.986508Z","iopub.status.idle":"2022-08-11T05:53:05.000399Z","shell.execute_reply.started":"2022-08-11T05:53:04.986466Z","shell.execute_reply":"2022-08-11T05:53:04.999382Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_test_new","metadata":{"execution":{"iopub.status.busy":"2022-08-11T05:53:05.002213Z","iopub.execute_input":"2022-08-11T05:53:05.002903Z","iopub.status.idle":"2022-08-11T05:53:05.042142Z","shell.execute_reply.started":"2022-08-11T05:53:05.002859Z","shell.execute_reply":"2022-08-11T05:53:05.040856Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Train-Test Split\n","metadata":{}},{"cell_type":"markdown","source":"### Splitting the data into Train and Test chunks for better evaluation","metadata":{}},{"cell_type":"code","source":"x_train,x_test,y_train,y_test = train_test_split(x,y,test_size = 0.2,random_state = 42)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T05:53:05.043915Z","iopub.execute_input":"2022-08-11T05:53:05.044637Z","iopub.status.idle":"2022-08-11T05:53:05.054932Z","shell.execute_reply.started":"2022-08-11T05:53:05.044581Z","shell.execute_reply":"2022-08-11T05:53:05.053494Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_train","metadata":{"execution":{"iopub.status.busy":"2022-08-11T05:53:05.056920Z","iopub.execute_input":"2022-08-11T05:53:05.057599Z","iopub.status.idle":"2022-08-11T05:53:05.087827Z","shell.execute_reply.started":"2022-08-11T05:53:05.057559Z","shell.execute_reply":"2022-08-11T05:53:05.086761Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_test","metadata":{"execution":{"iopub.status.busy":"2022-08-11T05:53:05.089115Z","iopub.execute_input":"2022-08-11T05:53:05.090230Z","iopub.status.idle":"2022-08-11T05:53:05.119835Z","shell.execute_reply.started":"2022-08-11T05:53:05.090197Z","shell.execute_reply":"2022-08-11T05:53:05.119034Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_train","metadata":{"execution":{"iopub.status.busy":"2022-08-11T05:53:05.123433Z","iopub.execute_input":"2022-08-11T05:53:05.125401Z","iopub.status.idle":"2022-08-11T05:53:05.134613Z","shell.execute_reply.started":"2022-08-11T05:53:05.125367Z","shell.execute_reply":"2022-08-11T05:53:05.133865Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_test","metadata":{"execution":{"iopub.status.busy":"2022-08-11T05:53:05.136372Z","iopub.execute_input":"2022-08-11T05:53:05.137116Z","iopub.status.idle":"2022-08-11T05:53:05.148139Z","shell.execute_reply.started":"2022-08-11T05:53:05.137071Z","shell.execute_reply":"2022-08-11T05:53:05.146754Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Defining several evaluation functions for convenience\n\n","metadata":{}},{"cell_type":"code","source":"def rmse_cv(model):\n    rmse = np.sqrt(-cross_val_score(model, x,y,scoring = \"neg_mean_squared_error\", cv = 5)).mean()\n    return rmse\n\ndef evaluation(y, predictions):\n    mae = mean_absolute_error(y,predictions)\n    mse = mean_squared_error(y,predictions)\n    rmse = np.sqrt(mean_squared_error(y,predictions))\n    r_squared = r2_score(y,predictions)\n    return mae,mse,rmse,r_squared","metadata":{"execution":{"iopub.status.busy":"2022-08-11T05:53:05.149987Z","iopub.execute_input":"2022-08-11T05:53:05.150826Z","iopub.status.idle":"2022-08-11T05:53:05.160152Z","shell.execute_reply.started":"2022-08-11T05:53:05.150787Z","shell.execute_reply":"2022-08-11T05:53:05.159043Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"models = pd.DataFrame(columns=[\"Model\",\"MAE\",\"MSE\",\"RMSE\",\"R2 Score\",\"RMSE (Cross-Validation)\"])\n","metadata":{"execution":{"iopub.status.busy":"2022-08-11T05:53:05.161712Z","iopub.execute_input":"2022-08-11T05:53:05.164028Z","iopub.status.idle":"2022-08-11T05:53:05.171779Z","shell.execute_reply.started":"2022-08-11T05:53:05.163932Z","shell.execute_reply":"2022-08-11T05:53:05.171002Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Machine Learning Models\n","metadata":{}},{"cell_type":"markdown","source":"### Linear Regression\n","metadata":{}},{"cell_type":"code","source":"lin_reg = LinearRegression()\nlin_reg.fit(x_train, y_train)\npredictions = lin_reg.predict(x_test)\n# print(predictions)\nmae, mse, rmse, r_squared = evaluation(y_test, predictions)\nprint(\"MAE:\", mae)\nprint(\"MSE:\", mse)\nprint(\"RMSE:\", rmse)\nprint(\"R2 Score:\", r_squared)\nprint(\"-\"*30)\nrmse_cross_val = rmse_cv(lin_reg)\nprint(\"RMSE Cross-Validation:\", rmse_cross_val)\n\nnew_row = {\"Model\": \"LinearRegression\",\"MAE\": mae, \"MSE\": mse, \"RMSE\": rmse, \"R2 Score\": r_squared, \"RMSE (Cross-Validation)\": rmse_cross_val}\nmodels = models.append(new_row, ignore_index=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T05:53:05.173117Z","iopub.execute_input":"2022-08-11T05:53:05.174159Z","iopub.status.idle":"2022-08-11T05:53:05.413127Z","shell.execute_reply.started":"2022-08-11T05:53:05.174118Z","shell.execute_reply":"2022-08-11T05:53:05.409717Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"models","metadata":{"execution":{"iopub.status.busy":"2022-08-11T05:53:05.415129Z","iopub.execute_input":"2022-08-11T05:53:05.415744Z","iopub.status.idle":"2022-08-11T05:53:05.449436Z","shell.execute_reply.started":"2022-08-11T05:53:05.415695Z","shell.execute_reply":"2022-08-11T05:53:05.447713Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Ridge Regression\n","metadata":{}},{"cell_type":"code","source":"ridge = Ridge()\nridge.fit(x_train,y_train)\npredictions = ridge.predict(x_test)\n\nmae, mse, rmse, r_squared = evaluation(y_test, predictions)\nprint(\"MAE:\", mae)\nprint(\"MSE:\", mse)\nprint(\"RMSE:\", rmse)\nprint(\"R2 Score:\", r_squared)\nprint(\"-\"*30)\nrmse_cross_val = rmse_cv(ridge)\nprint(\"RMSE Cross-Validation:\", rmse_cross_val)\n\nnew_row = {\"Model\": \"Ridge Regression\",\"MAE\": mae, \"MSE\": mse, \"RMSE\": rmse, \"R2 Score\": r_squared, \"RMSE (Cross-Validation)\": rmse_cross_val}\nmodels = models.append(new_row, ignore_index=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T05:53:05.455503Z","iopub.execute_input":"2022-08-11T05:53:05.456390Z","iopub.status.idle":"2022-08-11T05:53:05.643477Z","shell.execute_reply.started":"2022-08-11T05:53:05.456329Z","shell.execute_reply":"2022-08-11T05:53:05.642148Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"models","metadata":{"execution":{"iopub.status.busy":"2022-08-11T05:53:05.646373Z","iopub.execute_input":"2022-08-11T05:53:05.647306Z","iopub.status.idle":"2022-08-11T05:53:05.672708Z","shell.execute_reply.started":"2022-08-11T05:53:05.647263Z","shell.execute_reply":"2022-08-11T05:53:05.671280Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Lasso Regression","metadata":{}},{"cell_type":"code","source":"lasso  =  Lasso()\nlasso.fit(x_train,y_train)\npredictions = lasso.predict(x_test)\n\nmae, mse, rmse, r_squared = evaluation(y_test, predictions)\nprint(\"MAE:\", mae)\nprint(\"MSE:\", mse)\nprint(\"RMSE:\", rmse)\nprint(\"R2 Score:\", r_squared)\nprint(\"-\"*30)\nrmse_cross_val = rmse_cv(lasso)\nprint(\"RMSE Cross-Validation:\", rmse_cross_val)\n\nnew_row = {\"Model\": \"Lasso Regrssion\",\"MAE\": mae, \"MSE\": mse, \"RMSE\": rmse, \"R2 Score\": r_squared, \"RMSE (Cross-Validation)\": rmse_cross_val}\nmodels = models.append(new_row, ignore_index=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T05:53:05.678504Z","iopub.execute_input":"2022-08-11T05:53:05.679488Z","iopub.status.idle":"2022-08-11T05:53:06.246241Z","shell.execute_reply.started":"2022-08-11T05:53:05.679430Z","shell.execute_reply":"2022-08-11T05:53:06.244606Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Elastic Net\n","metadata":{}},{"cell_type":"code","source":"elastic_net = ElasticNet()\nelastic_net.fit(x_train,y_train)\npredictions = elastic_net.predict(x_test)\n\nmae, mse, rmse, r_squared = evaluation(y_test, predictions)\nprint(\"MAE:\", mae)\nprint(\"MSE:\", mse)\nprint(\"RMSE:\", rmse)\nprint(\"R2 Score:\", r_squared)\nprint(\"-\"*30)\nrmse_cross_val = rmse_cv(elastic_net)\nprint(\"RMSE Cross-Validation:\", rmse_cross_val)\n\nnew_row = {\"Model\": \"Elastic Net\",\"MAE\": mae, \"MSE\": mse, \"RMSE\": rmse, \"R2 Score\": r_squared, \"RMSE (Cross-Validation)\": rmse_cross_val}\nmodels = models.append(new_row, ignore_index=True)\n","metadata":{"execution":{"iopub.status.busy":"2022-08-11T05:53:06.248405Z","iopub.execute_input":"2022-08-11T05:53:06.250548Z","iopub.status.idle":"2022-08-11T05:53:06.426347Z","shell.execute_reply.started":"2022-08-11T05:53:06.250497Z","shell.execute_reply":"2022-08-11T05:53:06.424793Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Support Vector Machines\n","metadata":{}},{"cell_type":"code","source":"svr = SVR(C = 100000)\nsvr.fit(x_train,y_train)\npredictions = svr.predict(x_test)\n\nmae, mse, rmse, r_squared = evaluation(y_test, predictions)\nprint(\"MAE:\", mae)\nprint(\"MSE:\", mse)\nprint(\"RMSE:\", rmse)\nprint(\"R2 Score:\", r_squared)\nprint(\"-\"*30)\nrmse_cross_val = rmse_cv(svr)\nprint(\"RMSE Cross-Validation:\", rmse_cross_val)\n\nnew_row = {\"Model\": \"Support Vector MAchine\",\"MAE\": mae, \"MSE\": mse, \"RMSE\": rmse, \"R2 Score\": r_squared, \"RMSE (Cross-Validation)\": rmse_cross_val}\nmodels = models.append(new_row, ignore_index=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T05:53:06.428480Z","iopub.execute_input":"2022-08-11T05:53:06.430596Z","iopub.status.idle":"2022-08-11T05:53:08.024024Z","shell.execute_reply.started":"2022-08-11T05:53:06.430540Z","shell.execute_reply":"2022-08-11T05:53:08.022855Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Random Forest Regressor","metadata":{}},{"cell_type":"code","source":"random_forest = RandomForestRegressor(n_estimators = 100)\nrandom_forest.fit(x_train,y_train)\npredictions = random_forest.predict(x_test)\n\nmae, mse, rmse, r_squared = evaluation(y_test, predictions)\nprint(\"MAE:\", mae)\nprint(\"MSE:\", mse)\nprint(\"RMSE:\", rmse)\nprint(\"R2 Score:\", r_squared)\nprint(\"-\"*30)\nrmse_cross_val = rmse_cv(random_forest)\nprint(\"RMSE Cross-Validation:\", rmse_cross_val)\n\nnew_row = {\"Model\": \"Random Forest\",\"MAE\": mae, \"MSE\": mse, \"RMSE\": rmse, \"R2 Score\": r_squared, \"RMSE (Cross-Validation)\": rmse_cross_val}\nmodels = models.append(new_row, ignore_index=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T05:53:08.025766Z","iopub.execute_input":"2022-08-11T05:53:08.026238Z","iopub.status.idle":"2022-08-11T05:53:12.350825Z","shell.execute_reply.started":"2022-08-11T05:53:08.026206Z","shell.execute_reply":"2022-08-11T05:53:12.349731Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"models","metadata":{"execution":{"iopub.status.busy":"2022-08-11T05:53:12.352257Z","iopub.execute_input":"2022-08-11T05:53:12.353031Z","iopub.status.idle":"2022-08-11T05:53:12.369056Z","shell.execute_reply.started":"2022-08-11T05:53:12.352958Z","shell.execute_reply":"2022-08-11T05:53:12.368131Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### XGBoost Regressor\n","metadata":{}},{"cell_type":"code","source":"xgb = XGBRegressor(n_estimators =1000 , learning_rate = 0.01)\nxgb.fit(x_train,y_train)\npredictions = xgb.predict(x_test)\n# Print(predictions)\nmae, mse, rmse, r_squared = evaluation(y_test, predictions)\nprint(\"MAE:\", mae)\nprint(\"MSE:\", mse)\nprint(\"RMSE:\", rmse)\nprint(\"R2 Score:\", r_squared)\nprint(\"-\"*30)\nrmse_cross_val = rmse_cv(xgb)\nprint(\"RMSE Cross-Validation:\", rmse_cross_val)\n\nnew_row = {\"Model\": \"XGB\",\"MAE\": mae, \"MSE\": mse, \"RMSE\": rmse, \"R2 Score\": r_squared, \"RMSE (Cross-Validation)\": rmse_cross_val}\nmodels = models.append(new_row, ignore_index=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T05:53:12.370641Z","iopub.execute_input":"2022-08-11T05:53:12.371251Z","iopub.status.idle":"2022-08-11T05:53:36.876291Z","shell.execute_reply.started":"2022-08-11T05:53:12.371221Z","shell.execute_reply":"2022-08-11T05:53:36.875444Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Polynomial Regression (Degree=2)\n","metadata":{}},{"cell_type":"code","source":"poly_reg2 = PolynomialFeatures(degree = 2)\nx_train2 = poly_reg2.fit_transform(x_train)\nx_test2 = poly_reg2.transform(x_test)\n\nlin_reg_poly = LinearRegression()\nlin_reg_poly.fit(x_train2, y_train)\npredictions = lin_reg_poly.predict(x_test2)\n\nmae, mse, rmse, r_squared = evaluation(y_test, predictions)\nprint(\"MAE:\", mae)\nprint(\"MSE:\", mse)\nprint(\"RMSE:\", rmse)\nprint(\"R2 Score:\", r_squared)\nprint(\"-\"*30)\nrmse_cross_val = rmse_cv(lin_reg_poly)\nprint(\"RMSE Cross-Validation:\", rmse_cross_val)\n\nnew_row = {\"Model\": \"lin_reg_poly\",\"MAE\": mae, \"MSE\": mse, \"RMSE\": rmse, \"R2 Score\": r_squared, \"RMSE (Cross-Validation)\": rmse_cross_val}\nmodels = models.append(new_row, ignore_index=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T05:53:36.880154Z","iopub.execute_input":"2022-08-11T05:53:36.882181Z","iopub.status.idle":"2022-08-11T05:53:37.253694Z","shell.execute_reply.started":"2022-08-11T05:53:36.882146Z","shell.execute_reply":"2022-08-11T05:53:37.252085Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Model Comparison\n","metadata":{}},{"cell_type":"markdown","source":"### The less the Root Mean Squared Error (RMSE), The better the model is.","metadata":{}},{"cell_type":"code","source":"models","metadata":{"execution":{"iopub.status.busy":"2022-08-11T05:53:37.261071Z","iopub.execute_input":"2022-08-11T05:53:37.265896Z","iopub.status.idle":"2022-08-11T05:53:37.342349Z","shell.execute_reply.started":"2022-08-11T05:53:37.265834Z","shell.execute_reply":"2022-08-11T05:53:37.339494Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"models.sort_values(by=\"RMSE (Cross-Validation)\")","metadata":{"execution":{"iopub.status.busy":"2022-08-11T05:53:37.345946Z","iopub.execute_input":"2022-08-11T05:53:37.348669Z","iopub.status.idle":"2022-08-11T05:53:37.373614Z","shell.execute_reply.started":"2022-08-11T05:53:37.348622Z","shell.execute_reply":"2022-08-11T05:53:37.372819Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(12,8))\nsns.barplot(x=models[\"Model\"], y=models[\"RMSE (Cross-Validation)\"])\nplt.title(\"Models' RMSE Scores (Cross-Validated)\", size=15)\nplt.xticks(rotation=30, size=12)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-11T05:53:37.374932Z","iopub.execute_input":"2022-08-11T05:53:37.375292Z","iopub.status.idle":"2022-08-11T05:53:37.638234Z","shell.execute_reply.started":"2022-08-11T05:53:37.375261Z","shell.execute_reply":"2022-08-11T05:53:37.637263Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions_xgb_test = xgb.predict(x_test_new)\npredictions_xgb_test","metadata":{"execution":{"iopub.status.busy":"2022-08-11T05:53:37.639414Z","iopub.execute_input":"2022-08-11T05:53:37.639735Z","iopub.status.idle":"2022-08-11T05:53:37.669498Z","shell.execute_reply.started":"2022-08-11T05:53:37.639706Z","shell.execute_reply":"2022-08-11T05:53:37.668459Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"id_col_test = df_test['Id']\n\noutput_for_sub = pd.DataFrame(columns=[\"SalePrice\"])\n\noutput_for_sub.insert(0,'Id',id_col_test)\n\noutput_for_sub","metadata":{"execution":{"iopub.status.busy":"2022-08-11T05:53:37.670958Z","iopub.execute_input":"2022-08-11T05:53:37.671352Z","iopub.status.idle":"2022-08-11T05:53:37.690926Z","shell.execute_reply.started":"2022-08-11T05:53:37.671314Z","shell.execute_reply":"2022-08-11T05:53:37.690038Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"output_for_sub['SalePrice'] = predictions_xgb_test","metadata":{"execution":{"iopub.status.busy":"2022-08-11T05:53:37.692892Z","iopub.execute_input":"2022-08-11T05:53:37.694220Z","iopub.status.idle":"2022-08-11T05:53:37.700445Z","shell.execute_reply.started":"2022-08-11T05:53:37.694178Z","shell.execute_reply":"2022-08-11T05:53:37.699326Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"output_for_sub","metadata":{"execution":{"iopub.status.busy":"2022-08-11T05:53:37.702878Z","iopub.execute_input":"2022-08-11T05:53:37.704076Z","iopub.status.idle":"2022-08-11T05:53:37.718467Z","shell.execute_reply.started":"2022-08-11T05:53:37.704031Z","shell.execute_reply":"2022-08-11T05:53:37.717436Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"output_for_sub.to_csv('file1.csv')","metadata":{"execution":{"iopub.status.busy":"2022-08-11T05:53:37.719653Z","iopub.execute_input":"2022-08-11T05:53:37.720308Z","iopub.status.idle":"2022-08-11T05:53:37.732052Z","shell.execute_reply.started":"2022-08-11T05:53:37.720265Z","shell.execute_reply":"2022-08-11T05:53:37.731173Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}