{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nimport plotly.express as px\nimport plotly.io as pio\nimport scipy.stats as st\nfrom scipy import stats\nimport math\nimport missingno as msno\nfrom scipy.stats import norm, skew\nfrom collections import Counter\n%matplotlib inline\n\nfrom sklearn.preprocessing import LabelEncoder\nfrom sklearn.preprocessing import MinMaxScaler, RobustScaler, StandardScaler\nfrom sklearn.model_selection import train_test_split, GridSearchCV, cross_val_score, KFold\nfrom sklearn.metrics import mean_squared_error, mean_squared_log_error, r2_score\nfrom sklearn import model_selection\nfrom sklearn.pipeline import make_pipeline\n\nfrom xgboost import XGBRegressor\nfrom lightgbm import LGBMRegressor\nfrom sklearn.svm import SVR\nfrom sklearn.ensemble import RandomForestRegressor, GradientBoostingRegressor\nfrom sklearn.linear_model import Ridge, RidgeCV, Lasso, LassoCV\nfrom mlxtend.regressor import StackingCVRegressor\n\nimport plotly.offline as pof\npof.init_notebook_mode()\n\n# to ignore warnings\nimport warnings\nwarnings.filterwarnings(\"ignore\")\n\n#to see model hyperparameters\nfrom sklearn import set_config\nset_config(print_changed_only = False)\n\n# to show all columns\npd.set_option('display.max_columns', 82)","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-27T11:05:04.112839Z","iopub.execute_input":"2022-07-27T11:05:04.113287Z","iopub.status.idle":"2022-07-27T11:05:04.188305Z","shell.execute_reply.started":"2022-07-27T11:05:04.113252Z","shell.execute_reply":"2022-07-27T11:05:04.186918Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.read_csv(\"../input/house-prices-advanced-regression-techniques/train.csv\")\ntest_df = pd.read_csv(\"../input/house-prices-advanced-regression-techniques/test.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-07-27T11:05:04.299588Z","iopub.execute_input":"2022-07-27T11:05:04.300335Z","iopub.status.idle":"2022-07-27T11:05:04.341820Z","shell.execute_reply.started":"2022-07-27T11:05:04.300279Z","shell.execute_reply":"2022-07-27T11:05:04.340826Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-27T11:05:04.344002Z","iopub.execute_input":"2022-07-27T11:05:04.344749Z","iopub.status.idle":"2022-07-27T11:05:04.402218Z","shell.execute_reply.started":"2022-07-27T11:05:04.344690Z","shell.execute_reply":"2022-07-27T11:05:04.401073Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"The number of rows in train data is {0}, and the number of columns in train data is {1}\".\n      format(train_df.shape[0], train_df.shape[1]))\n      \nprint(\"The number of rows in test data is {0}, and the number of columns in test data is {1}\".\n      format(test_df.shape[0], test_df.shape[1]))","metadata":{"execution":{"iopub.status.busy":"2022-07-27T11:05:04.404757Z","iopub.execute_input":"2022-07-27T11:05:04.405327Z","iopub.status.idle":"2022-07-27T11:05:04.417555Z","shell.execute_reply.started":"2022-07-27T11:05:04.405287Z","shell.execute_reply":"2022-07-27T11:05:04.416192Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-27T11:05:04.444216Z","iopub.execute_input":"2022-07-27T11:05:04.445389Z","iopub.status.idle":"2022-07-27T11:05:04.470967Z","shell.execute_reply.started":"2022-07-27T11:05:04.445328Z","shell.execute_reply":"2022-07-27T11:05:04.469637Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.describe()","metadata":{"execution":{"iopub.status.busy":"2022-07-27T11:05:04.497906Z","iopub.execute_input":"2022-07-27T11:05:04.498569Z","iopub.status.idle":"2022-07-27T11:05:04.629023Z","shell.execute_reply.started":"2022-07-27T11:05:04.498523Z","shell.execute_reply":"2022-07-27T11:05:04.627758Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"numeric_vars = train_df.select_dtypes(\"number\")\n\ndef diagnostic(df, var):\n    fig = plt.figure(figsize = (17, 6))\n    plt.subplot(1,3,1)\n    df[var].hist(bins = 40)\n    plt.title(\"Distribution of {}\".format(var))\n    \n    plt.subplot(1,3,2)\n    stats.probplot(df[var], dist = \"norm\", plot = plt)\n    plt.ylabel(\"Quantiles\")\n    \n    plt.subplot(1,3,3)\n    sns.boxplot(y = df[var])\n    plt.title(\"Boxplot\")\n    plt.show()\n    \nfor var in numeric_vars:\n    diagnostic(train_df, var)   ","metadata":{"execution":{"iopub.status.busy":"2022-07-27T11:05:04.632268Z","iopub.execute_input":"2022-07-27T11:05:04.632766Z","iopub.status.idle":"2022-07-27T11:05:27.341209Z","shell.execute_reply.started":"2022-07-27T11:05:04.632717Z","shell.execute_reply":"2022-07-27T11:05:27.340037Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.hist(figsize = (30, 30), bins = 20, legend = False)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-27T11:05:27.342745Z","iopub.execute_input":"2022-07-27T11:05:27.343145Z","iopub.status.idle":"2022-07-27T11:05:34.726474Z","shell.execute_reply.started":"2022-07-27T11:05:27.343114Z","shell.execute_reply":"2022-07-27T11:05:34.725332Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.catplot(x = \"HouseStyle\",\n            y = \"SalePrice\",\n            kind = \"box\",\n            height = 6,\n            aspect = 1.8,\n            color = \"#FBC02D\",\n            data = train_df).set(title = \"Sale prices of the houses by house styles\");","metadata":{"execution":{"iopub.status.busy":"2022-07-27T11:05:34.729296Z","iopub.execute_input":"2022-07-27T11:05:34.730112Z","iopub.status.idle":"2022-07-27T11:05:35.230174Z","shell.execute_reply.started":"2022-07-27T11:05:34.730066Z","shell.execute_reply":"2022-07-27T11:05:35.228915Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.catplot(x = \"Neighborhood\",\n            y = \"SalePrice\",\n            kind = \"box\",\n            height = 10,\n            aspect = 2,\n            color = \"#00FF00\",\n            data = train_df).set(title = \"Sale prices of the houses by neighborhood\");","metadata":{"execution":{"iopub.status.busy":"2022-07-27T11:05:35.231528Z","iopub.execute_input":"2022-07-27T11:05:35.231921Z","iopub.status.idle":"2022-07-27T11:05:36.105770Z","shell.execute_reply.started":"2022-07-27T11:05:35.231880Z","shell.execute_reply":"2022-07-27T11:05:36.104901Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.catplot(x = \"RoofStyle\",\n            y = \"SalePrice\",\n            kind = \"strip\",\n            height = 7,\n            aspect = 2,\n            color = \"#F415DC\",\n            data = train_df).set(title = \"Sale prices of the houses by roof styles\");","metadata":{"execution":{"iopub.status.busy":"2022-07-27T11:05:36.107032Z","iopub.execute_input":"2022-07-27T11:05:36.108176Z","iopub.status.idle":"2022-07-27T11:05:36.507839Z","shell.execute_reply.started":"2022-07-27T11:05:36.108132Z","shell.execute_reply":"2022-07-27T11:05:36.506968Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.catplot(x = \"Condition1\",\n            y = \"SalePrice\",\n            kind = \"violin\",\n            height = 7,\n            aspect = 2.5,\n            color = \"#B3E5FC\",\n            data = train_df).set(title = \"Sale prices of the houses by condition one\");","metadata":{"execution":{"iopub.status.busy":"2022-07-27T11:05:36.509326Z","iopub.execute_input":"2022-07-27T11:05:36.509952Z","iopub.status.idle":"2022-07-27T11:05:37.079492Z","shell.execute_reply.started":"2022-07-27T11:05:36.509910Z","shell.execute_reply":"2022-07-27T11:05:37.078261Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.catplot(x = \"Condition2\",\n            y = \"SalePrice\",\n            kind = \"swarm\",\n            height = 7,\n            aspect = 2.5,\n            color = \"#80202B\",\n            data = train_df).set(title = \"Sale prices of the houses by condition two\");","metadata":{"execution":{"iopub.status.busy":"2022-07-27T11:05:37.082844Z","iopub.execute_input":"2022-07-27T11:05:37.083598Z","iopub.status.idle":"2022-07-27T11:05:40.167029Z","shell.execute_reply.started":"2022-07-27T11:05:37.083549Z","shell.execute_reply":"2022-07-27T11:05:40.165795Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.catplot(x = \"Condition1\",\n            y = \"SalePrice\",\n            kind = \"swarm\",\n            height = 7,\n            aspect = 2.5,\n            color = \"#000000\",\n            data = train_df).set(title = \"Sale prices of the houses by condition one\");","metadata":{"execution":{"iopub.status.busy":"2022-07-27T11:05:40.172671Z","iopub.execute_input":"2022-07-27T11:05:40.173058Z","iopub.status.idle":"2022-07-27T11:05:42.467052Z","shell.execute_reply.started":"2022-07-27T11:05:40.173024Z","shell.execute_reply":"2022-07-27T11:05:42.465776Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sp = train_df[\"SalePrice\"]\noq = train_df[\"OverallQual\"]\ndf = pd.concat([sp, oq], axis=1)      \nf, ax = plt.subplots(figsize = (18, 11))\nfig = sns.boxplot(df[\"OverallQual\"], df[\"SalePrice\"]);","metadata":{"execution":{"iopub.status.busy":"2022-07-27T11:05:42.468362Z","iopub.execute_input":"2022-07-27T11:05:42.469352Z","iopub.status.idle":"2022-07-27T11:05:42.802139Z","shell.execute_reply.started":"2022-07-27T11:05:42.469311Z","shell.execute_reply":"2022-07-27T11:05:42.800998Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, axes = plt.subplots(2, 2, figsize = (15, 10))\naxes = axes.flatten()\n\nsns.barplot(ax = axes[0],\n            x = train_df[\"HouseStyle\"].value_counts().index,\n            y = train_df[\"HouseStyle\"].value_counts(),\n            saturation = 1).set(title = \"Frequency of classes of the 'HouseStyle' variable\");\n\nsns.barplot(ax = axes[1],\n            x = train_df[\"Street\"].value_counts().index,\n            y = train_df[\"Street\"].value_counts(),\n            saturation = 1).set(title = \"Frequency of cases of the 'street' variable\");\n\nsns.barplot(ax = axes[2],\n            x = train_df[\"Condition1\"].value_counts().index,\n            y = train_df[\"Condition1\"].value_counts(),\n            saturation = 1).set(title = \"Frequency of cases of the 'condition1' variable\");\n\nsns.barplot(ax = axes[3],\n            x = train_df[\"Condition2\"].value_counts().index,\n            y = train_df[\"Condition2\"].value_counts(),\n            saturation = 1).set(title = \"Frequency of classes of the 'condition2' variable\");","metadata":{"execution":{"iopub.status.busy":"2022-07-27T11:05:42.803667Z","iopub.execute_input":"2022-07-27T11:05:42.804188Z","iopub.status.idle":"2022-07-27T11:05:43.474059Z","shell.execute_reply.started":"2022-07-27T11:05:42.804142Z","shell.execute_reply":"2022-07-27T11:05:43.472636Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.set_theme(style = \"darkgrid\")\n\ng = sns.JointGrid(data = train_df, size = 7, height = 5, x = \"LotArea\", y = \"SalePrice\", space = 0.5)\ng.plot_joint(sns.kdeplot, fill = True, thresh = 0, cmap = \"bone\")\ng.plot_marginals(sns.histplot, color = \"#80202B\", alpha = 1, bins = 30);","metadata":{"execution":{"iopub.status.busy":"2022-07-27T11:05:43.475821Z","iopub.execute_input":"2022-07-27T11:05:43.476232Z","iopub.status.idle":"2022-07-27T11:05:45.290982Z","shell.execute_reply.started":"2022-07-27T11:05:43.476199Z","shell.execute_reply":"2022-07-27T11:05:45.289784Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.set_theme(style = \"whitegrid\")\n\ng = sns.JointGrid(data = train_df, size = 7, height = 5, x = \"GrLivArea\", y = \"SalePrice\", space = 0.5)\ng.plot_joint(sns.kdeplot, fill = False, thresh = 0, cmap = \"summer\")\ng.plot_marginals(sns.histplot, color = \"#808080\", alpha = 1, bins = 30);","metadata":{"execution":{"iopub.status.busy":"2022-07-27T11:05:45.292663Z","iopub.execute_input":"2022-07-27T11:05:45.293552Z","iopub.status.idle":"2022-07-27T11:05:47.039379Z","shell.execute_reply.started":"2022-07-27T11:05:45.293509Z","shell.execute_reply":"2022-07-27T11:05:47.038027Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"g = sns.JointGrid(data = train_df, size = 7, height = 5, x = \"OverallQual\", y = \"SalePrice\", space = 0.5)\ng.plot_joint(sns.kdeplot, fill = True, thresh = 0, cmap = \"binary\")\ng.plot_marginals(sns.histplot, color = \"#808080\", alpha = 1, bins = 30);","metadata":{"execution":{"iopub.status.busy":"2022-07-27T11:05:47.041333Z","iopub.execute_input":"2022-07-27T11:05:47.041777Z","iopub.status.idle":"2022-07-27T11:05:48.838922Z","shell.execute_reply.started":"2022-07-27T11:05:47.041733Z","shell.execute_reply":"2022-07-27T11:05:48.837440Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = px.histogram(train_df, x = \"OverallQual\",\n                   y = \"SalePrice\",\n                   marginal = None,\n                   color = \"HouseStyle\",\n                   text_auto = True,\n                   hover_data  = train_df.columns,\n                   height = 500, width = 800)\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-27T11:05:48.840599Z","iopub.execute_input":"2022-07-27T11:05:48.841004Z","iopub.status.idle":"2022-07-27T11:05:48.983097Z","shell.execute_reply.started":"2022-07-27T11:05:48.840968Z","shell.execute_reply":"2022-07-27T11:05:48.981649Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = px.histogram(train_df, x = \"BedroomAbvGr\",\n                   y = \"SalePrice\",\n                   marginal = None,\n                   color = \"HouseStyle\", text_auto = True,\n                   hover_data  = train_df.columns,\n                   height = 500, width = 800)\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-27T11:05:48.985089Z","iopub.execute_input":"2022-07-27T11:05:48.985537Z","iopub.status.idle":"2022-07-27T11:05:49.129306Z","shell.execute_reply.started":"2022-07-27T11:05:48.985488Z","shell.execute_reply":"2022-07-27T11:05:49.128155Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = px.histogram(train_df, x = \"RoofMatl\",\n                   y = \"SalePrice\",\n                   marginal = None,\n                   color = \"Heating\",\n                   text_auto = True,\n                   hover_data  = train_df.columns,\n                   height = 500, width = 800)\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-27T11:05:49.130933Z","iopub.execute_input":"2022-07-27T11:05:49.131380Z","iopub.status.idle":"2022-07-27T11:05:49.276560Z","shell.execute_reply.started":"2022-07-27T11:05:49.131334Z","shell.execute_reply":"2022-07-27T11:05:49.275230Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = px.ecdf(train_df, x = \"SalePrice\", log_x = True, log_y = True,\n              color = \"Street\", height = 500, width = 800)\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-27T11:05:49.277981Z","iopub.execute_input":"2022-07-27T11:05:49.278470Z","iopub.status.idle":"2022-07-27T11:05:49.354193Z","shell.execute_reply.started":"2022-07-27T11:05:49.278429Z","shell.execute_reply":"2022-07-27T11:05:49.352951Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = px.ecdf(train_df, x = \"SalePrice\", log_x = True, log_y = True,\n              color = \"GarageCars\", height = 500, width = 800)\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-27T11:05:49.355828Z","iopub.execute_input":"2022-07-27T11:05:49.356205Z","iopub.status.idle":"2022-07-27T11:05:49.912041Z","shell.execute_reply.started":"2022-07-27T11:05:49.356171Z","shell.execute_reply":"2022-07-27T11:05:49.911086Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = px.ecdf(train_df, x = \"SalePrice\", log_x = True, log_y = True,\n              color = \"Heating\", height = 500, width = 800)\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-27T11:05:49.913584Z","iopub.execute_input":"2022-07-27T11:05:49.914233Z","iopub.status.idle":"2022-07-27T11:05:50.011083Z","shell.execute_reply.started":"2022-07-27T11:05:49.914186Z","shell.execute_reply":"2022-07-27T11:05:50.010110Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = px.bar(train_df, x = \"HouseStyle\", y = \"SalePrice\", color = \"SaleType\",\n             pattern_shape = \"Street\", pattern_shape_sequence=[\"x\", \"+\"],\n             text_auto = True, height = 500, width = 830)\nfig.show()\n","metadata":{"execution":{"iopub.status.busy":"2022-07-27T11:05:50.012700Z","iopub.execute_input":"2022-07-27T11:05:50.013089Z","iopub.status.idle":"2022-07-27T11:05:50.137317Z","shell.execute_reply.started":"2022-07-27T11:05:50.013055Z","shell.execute_reply":"2022-07-27T11:05:50.135912Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = px.density_heatmap(train_df, x = \"LotArea\", y = \"SalePrice\", marginal_x = \"rug\",\n                         marginal_y = \"histogram\", height = 500, width = 830, text_auto = True,\n                         title = \"Density heatmap between LotArea and SalePrice variables\")\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-27T11:05:50.139460Z","iopub.execute_input":"2022-07-27T11:05:50.140524Z","iopub.status.idle":"2022-07-27T11:05:50.270672Z","shell.execute_reply.started":"2022-07-27T11:05:50.140472Z","shell.execute_reply":"2022-07-27T11:05:50.269558Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = px.density_heatmap(train_df, x = \"YearBuilt\", y = \"SalePrice\", marginal_x = \"rug\",\n                         marginal_y = \"histogram\", height = 500, width = 830, text_auto = True,\n                         title = \"Density heatmap between YearBuilt and SalePrice variables\")\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-27T11:05:50.272403Z","iopub.execute_input":"2022-07-27T11:05:50.272939Z","iopub.status.idle":"2022-07-27T11:05:50.399351Z","shell.execute_reply.started":"2022-07-27T11:05:50.272898Z","shell.execute_reply":"2022-07-27T11:05:50.398189Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = px.density_heatmap(train_df, x = \"GrLivArea\", y = \"SalePrice\", marginal_x = \"rug\",\n                         marginal_y = \"histogram\", height = 500, width = 830, text_auto = True,\n                         title = \"Density heatmap between GrLivArea and SalePrice variables\")\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-27T11:05:50.405522Z","iopub.execute_input":"2022-07-27T11:05:50.405928Z","iopub.status.idle":"2022-07-27T11:05:50.533566Z","shell.execute_reply.started":"2022-07-27T11:05:50.405885Z","shell.execute_reply":"2022-07-27T11:05:50.532091Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots(figsize = (12, 6))\nsale_price = list()\nfor sp in train_df[\"SalePrice\"].values:\n    sale_price.append(sp)\nsale_price = pd.Series(sale_price)\nsale_price.plot(kind = \"line\", colormap = \"autumn\").set_title(\"Sale prices of houses\");","metadata":{"execution":{"iopub.status.busy":"2022-07-27T11:05:50.535051Z","iopub.execute_input":"2022-07-27T11:05:50.535428Z","iopub.status.idle":"2022-07-27T11:05:50.826971Z","shell.execute_reply.started":"2022-07-27T11:05:50.535395Z","shell.execute_reply":"2022-07-27T11:05:50.825815Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots(figsize = (12, 6))\nyear_built = list()\nfor year in train_df[\"YearBuilt\"].values:\n    year_built.append(year)\nyear_built = pd.Series(year_built)\nyear_built.plot(kind = \"line\", colormap = \"summer\").set_title(\"Building years of houses\");","metadata":{"execution":{"iopub.status.busy":"2022-07-27T11:05:50.828438Z","iopub.execute_input":"2022-07-27T11:05:50.829458Z","iopub.status.idle":"2022-07-27T11:05:51.136344Z","shell.execute_reply.started":"2022-07-27T11:05:50.829409Z","shell.execute_reply":"2022-07-27T11:05:51.135027Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Basic descriptive statistics of the target variable - 'SalePrice': \\n\\n\",\n      train_df[\"SalePrice\"].describe())","metadata":{"execution":{"iopub.status.busy":"2022-07-27T11:05:51.138191Z","iopub.execute_input":"2022-07-27T11:05:51.139223Z","iopub.status.idle":"2022-07-27T11:05:51.150304Z","shell.execute_reply.started":"2022-07-27T11:05:51.139170Z","shell.execute_reply":"2022-07-27T11:05:51.148942Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df[\"YearBuilt\"].describe()","metadata":{"execution":{"iopub.status.busy":"2022-07-27T11:05:51.152179Z","iopub.execute_input":"2022-07-27T11:05:51.153114Z","iopub.status.idle":"2022-07-27T11:05:51.170443Z","shell.execute_reply.started":"2022-07-27T11:05:51.153064Z","shell.execute_reply":"2022-07-27T11:05:51.169420Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Skewness: \", train_df[\"SalePrice\"].skew())","metadata":{"execution":{"iopub.status.busy":"2022-07-27T11:05:51.171880Z","iopub.execute_input":"2022-07-27T11:05:51.173052Z","iopub.status.idle":"2022-07-27T11:05:51.186093Z","shell.execute_reply.started":"2022-07-27T11:05:51.173005Z","shell.execute_reply":"2022-07-27T11:05:51.185158Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Kurtosis: \", train_df[\"SalePrice\"].kurt())","metadata":{"execution":{"iopub.status.busy":"2022-07-27T11:05:51.187413Z","iopub.execute_input":"2022-07-27T11:05:51.188290Z","iopub.status.idle":"2022-07-27T11:05:51.200057Z","shell.execute_reply.started":"2022-07-27T11:05:51.188250Z","shell.execute_reply":"2022-07-27T11:05:51.199193Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.set(rc = {\"figure.figsize\" : (11, 7)})\nsns.distplot(train_df[\"SalePrice\"], color = \"black\")","metadata":{"execution":{"iopub.status.busy":"2022-07-27T11:05:51.201524Z","iopub.execute_input":"2022-07-27T11:05:51.202171Z","iopub.status.idle":"2022-07-27T11:05:51.561005Z","shell.execute_reply.started":"2022-07-27T11:05:51.202125Z","shell.execute_reply":"2022-07-27T11:05:51.559899Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# we are fixing the variable distribution\ntrain_df[\"SalePrice\"] = np.log1p(train_df[\"SalePrice\"])\ntrain_df[\"SalePrice\"].head(n = 10)","metadata":{"execution":{"iopub.status.busy":"2022-07-27T11:05:51.562737Z","iopub.execute_input":"2022-07-27T11:05:51.563183Z","iopub.status.idle":"2022-07-27T11:05:51.575100Z","shell.execute_reply.started":"2022-07-27T11:05:51.563138Z","shell.execute_reply":"2022-07-27T11:05:51.573940Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize = [40, 20], facecolor = \"#F7F4F4\")\nsns.heatmap(train_df.corr(), annot = True, cmap = \"summer\");","metadata":{"execution":{"iopub.status.busy":"2022-07-27T11:05:51.576755Z","iopub.execute_input":"2022-07-27T11:05:51.577247Z","iopub.status.idle":"2022-07-27T11:05:57.751398Z","shell.execute_reply.started":"2022-07-27T11:05:51.577202Z","shell.execute_reply":"2022-07-27T11:05:57.749984Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"correlation_df = train_df.corr()\nhigh_correlation_variables = correlation_df.index[abs(correlation_df[\"SalePrice\"]) > 0.4]\nhigh_correlation_variables","metadata":{"execution":{"iopub.status.busy":"2022-07-27T11:05:57.752977Z","iopub.execute_input":"2022-07-27T11:05:57.754078Z","iopub.status.idle":"2022-07-27T11:05:57.772562Z","shell.execute_reply.started":"2022-07-27T11:05:57.754031Z","shell.execute_reply":"2022-07-27T11:05:57.771448Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize = (25, 12))\nsns.heatmap(train_df[high_correlation_variables].corr(), annot = True, cmap = \"binary\");","metadata":{"execution":{"iopub.status.busy":"2022-07-27T11:05:57.774060Z","iopub.execute_input":"2022-07-27T11:05:57.774466Z","iopub.status.idle":"2022-07-27T11:05:58.900482Z","shell.execute_reply.started":"2022-07-27T11:05:57.774432Z","shell.execute_reply":"2022-07-27T11:05:58.899180Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = px.imshow(train_df[high_correlation_variables])\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-27T11:05:58.902028Z","iopub.execute_input":"2022-07-27T11:05:58.902516Z","iopub.status.idle":"2022-07-27T11:05:58.974622Z","shell.execute_reply.started":"2022-07-27T11:05:58.902470Z","shell.execute_reply":"2022-07-27T11:05:58.973222Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.pairplot(train_df[high_correlation_variables], corner = True);","metadata":{"execution":{"iopub.status.busy":"2022-07-27T11:05:58.976153Z","iopub.execute_input":"2022-07-27T11:05:58.976593Z","iopub.status.idle":"2022-07-27T11:06:21.564569Z","shell.execute_reply.started":"2022-07-27T11:05:58.976560Z","shell.execute_reply":"2022-07-27T11:06:21.563409Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df[high_correlation_variables].corr()","metadata":{"execution":{"iopub.status.busy":"2022-07-27T11:06:21.566238Z","iopub.execute_input":"2022-07-27T11:06:21.566741Z","iopub.status.idle":"2022-07-27T11:06:21.601712Z","shell.execute_reply.started":"2022-07-27T11:06:21.566698Z","shell.execute_reply":"2022-07-27T11:06:21.600929Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# variables that have highest correlation with target variable \n\nvariables = [\"OverallQual\", \"GrLivArea\", \"GarageCars\", \"TotalBsmtSF\", \"FullBath\", \"YearBuilt\"]\nfor var in variables:\n    print(\"Correlation coefficient:\", train_df[\"SalePrice\"].corr(train_df[var]))","metadata":{"execution":{"iopub.status.busy":"2022-07-27T11:06:21.603245Z","iopub.execute_input":"2022-07-27T11:06:21.604461Z","iopub.status.idle":"2022-07-27T11:06:21.617277Z","shell.execute_reply.started":"2022-07-27T11:06:21.604413Z","shell.execute_reply":"2022-07-27T11:06:21.615800Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"variables = [\"SalePrice\", \"OverallQual\", \"GrLivArea\", \"GarageCars\", \"TotalBsmtSF\", \"FullBath\", \"YearBuilt\"]\nsns.pairplot(train_df[variables]);","metadata":{"execution":{"iopub.status.busy":"2022-07-27T11:06:21.618894Z","iopub.execute_input":"2022-07-27T11:06:21.619734Z","iopub.status.idle":"2022-07-27T11:06:31.457386Z","shell.execute_reply.started":"2022-07-27T11:06:21.619697Z","shell.execute_reply":"2022-07-27T11:06:31.456188Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# ****Dropping Outliers****","metadata":{}},{"cell_type":"code","source":"def outlier_detection_train(df, n, columns):\n    rows = []\n    will_drop_train = []\n    for col in columns:\n        Q1 = np.nanpercentile(df[col], 25)\n        Q3 = np.nanpercentile(df[col], 75)\n        IQR = Q3 - Q1\n        outlier_point = 1.5 * IQR\n        rows.extend(df[(df[col] < Q1 - outlier_point)|(df[col] > Q3 + outlier_point)].index)\n    for r, c in Counter(rows).items():\n        if c >= n: will_drop_train.append(r)\n    return will_drop_train\n\nwill_drop_train = outlier_detection_train(train_df, 5, train_df.select_dtypes([\"float\", \"int\"]).columns)\ntrain_df.drop(will_drop_train, inplace = True, axis = 0)","metadata":{"execution":{"iopub.status.busy":"2022-07-27T11:06:31.458841Z","iopub.execute_input":"2022-07-27T11:06:31.459277Z","iopub.status.idle":"2022-07-27T11:06:31.522254Z","shell.execute_reply.started":"2022-07-27T11:06:31.459245Z","shell.execute_reply":"2022-07-27T11:06:31.520813Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Null Values**","metadata":{}},{"cell_type":"code","source":"y_train = train_df[\"SalePrice\"]","metadata":{"execution":{"iopub.status.busy":"2022-07-27T11:06:31.523981Z","iopub.execute_input":"2022-07-27T11:06:31.524453Z","iopub.status.idle":"2022-07-27T11:06:31.530889Z","shell.execute_reply.started":"2022-07-27T11:06:31.524406Z","shell.execute_reply":"2022-07-27T11:06:31.529439Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# combine train and test data for convenience\n\ntrain_and_test_df = pd.concat([train_df, test_df], axis = 0)\ntrain_and_test_df = train_and_test_df.drop([\"Id\", \"SalePrice\"], axis = 1)","metadata":{"execution":{"iopub.status.busy":"2022-07-27T11:06:31.532334Z","iopub.execute_input":"2022-07-27T11:06:31.532777Z","iopub.status.idle":"2022-07-27T11:06:31.567030Z","shell.execute_reply.started":"2022-07-27T11:06:31.532740Z","shell.execute_reply":"2022-07-27T11:06:31.565816Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# create dataframes consist of total number and percent of missing data\n\nnumber_of_missing_df = train_and_test_df.isnull().sum().sort_values(ascending = False)\npercent_of_missing_df = ((train_and_test_df.isnull().sum() / train_and_test_df.isnull().count())*100).sort_values(ascending = False)","metadata":{"execution":{"iopub.status.busy":"2022-07-27T11:06:31.568446Z","iopub.execute_input":"2022-07-27T11:06:31.568779Z","iopub.status.idle":"2022-07-27T11:06:31.600737Z","shell.execute_reply.started":"2022-07-27T11:06:31.568749Z","shell.execute_reply":"2022-07-27T11:06:31.599301Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# combine the dataframes and print\n\nmissing_df = pd.concat([number_of_missing_df,\n                        percent_of_missing_df],\n                        keys = [\"total number of missing data\", 'total percent of missing data'],\n                        axis = 1)\n\n\nprint(missing_df.head(20))","metadata":{"execution":{"iopub.status.busy":"2022-07-27T11:06:31.602900Z","iopub.execute_input":"2022-07-27T11:06:31.603558Z","iopub.status.idle":"2022-07-27T11:06:31.614566Z","shell.execute_reply.started":"2022-07-27T11:06:31.603518Z","shell.execute_reply":"2022-07-27T11:06:31.613070Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"msno.matrix(train_and_test_df);","metadata":{"execution":{"iopub.status.busy":"2022-07-27T11:06:31.616178Z","iopub.execute_input":"2022-07-27T11:06:31.617220Z","iopub.status.idle":"2022-07-27T11:06:32.141399Z","shell.execute_reply.started":"2022-07-27T11:06:31.617170Z","shell.execute_reply":"2022-07-27T11:06:32.140084Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_and_test_df = train_and_test_df.drop((missing_df[missing_df[\"total number of missing data\"] > 100]).index, axis = 1)\ntrain_and_test_df.isnull().sum().sort_values(ascending = False)","metadata":{"execution":{"iopub.status.busy":"2022-07-27T11:06:32.142758Z","iopub.execute_input":"2022-07-27T11:06:32.143128Z","iopub.status.idle":"2022-07-27T11:06:32.164544Z","shell.execute_reply.started":"2022-07-27T11:06:32.143095Z","shell.execute_reply":"2022-07-27T11:06:32.163225Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"numeric_data = [column for column in train_and_test_df.select_dtypes([\"int\", \"float\"])]\ncategoric_data = [column for column in train_and_test_df.select_dtypes(exclude = [\"int\", \"float\"])]\n\nfor col in numeric_data:\n    train_and_test_df[col].fillna(train_and_test_df[col].median(), inplace = True)\n        \nfor col in categoric_data:\n    train_and_test_df[col].fillna(train_and_test_df[col].value_counts().index[0], inplace = True)","metadata":{"execution":{"iopub.status.busy":"2022-07-27T11:06:32.166254Z","iopub.execute_input":"2022-07-27T11:06:32.166772Z","iopub.status.idle":"2022-07-27T11:06:32.235704Z","shell.execute_reply.started":"2022-07-27T11:06:32.166723Z","shell.execute_reply":"2022-07-27T11:06:32.234456Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#ckeck whether there are missing values\n\ntrain_and_test_df.isnull().sum().sum()","metadata":{"execution":{"iopub.status.busy":"2022-07-27T11:06:32.237733Z","iopub.execute_input":"2022-07-27T11:06:32.238384Z","iopub.status.idle":"2022-07-27T11:06:32.253055Z","shell.execute_reply.started":"2022-07-27T11:06:32.238338Z","shell.execute_reply":"2022-07-27T11:06:32.251563Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# we select numeric variables of the dataset\nnumeric_data = [column for column in train_and_test_df.select_dtypes([\"int\", \"float\"])]\n\n# we check skew degree of that variables\nvars_skewed = train_and_test_df[numeric_data].apply(lambda x: skew(x)).sort_values()","metadata":{"execution":{"iopub.status.busy":"2022-07-27T11:06:32.255488Z","iopub.execute_input":"2022-07-27T11:06:32.256982Z","iopub.status.idle":"2022-07-27T11:06:32.280057Z","shell.execute_reply.started":"2022-07-27T11:06:32.256938Z","shell.execute_reply":"2022-07-27T11:06:32.279041Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# we fix skew with 'log1p' function of numpy\n\nfor var in vars_skewed.index:\n    train_and_test_df[var] = np.log1p(train_and_test_df[var])","metadata":{"execution":{"iopub.status.busy":"2022-07-27T11:06:32.281401Z","iopub.execute_input":"2022-07-27T11:06:32.282735Z","iopub.status.idle":"2022-07-27T11:06:32.307524Z","shell.execute_reply.started":"2022-07-27T11:06:32.282687Z","shell.execute_reply":"2022-07-27T11:06:32.306229Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_and_test_df = train_and_test_df.drop([\"GarageArea\", \"1stFlrSF\", \"TotRmsAbvGrd\"],\n                                          axis = 1)","metadata":{"execution":{"iopub.status.busy":"2022-07-27T11:06:32.309208Z","iopub.execute_input":"2022-07-27T11:06:32.309539Z","iopub.status.idle":"2022-07-27T11:06:32.317042Z","shell.execute_reply.started":"2022-07-27T11:06:32.309508Z","shell.execute_reply":"2022-07-27T11:06:32.316051Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_and_test_df = pd.get_dummies(train_and_test_df, drop_first = True)\ntrain_and_test_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-27T11:06:59.062486Z","iopub.execute_input":"2022-07-27T11:06:59.062887Z","iopub.status.idle":"2022-07-27T11:06:59.163260Z","shell.execute_reply.started":"2022-07-27T11:06:59.062832Z","shell.execute_reply":"2022-07-27T11:06:59.162115Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_train = train_and_test_df[:len(train_df)]\nx_test = train_and_test_df[len(train_df):]","metadata":{"execution":{"iopub.status.busy":"2022-07-27T11:07:10.944539Z","iopub.execute_input":"2022-07-27T11:07:10.944960Z","iopub.status.idle":"2022-07-27T11:07:10.951343Z","shell.execute_reply.started":"2022-07-27T11:07:10.944925Z","shell.execute_reply":"2022-07-27T11:07:10.950068Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_train.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-27T11:07:25.556139Z","iopub.execute_input":"2022-07-27T11:07:25.557153Z","iopub.status.idle":"2022-07-27T11:07:25.618159Z","shell.execute_reply.started":"2022-07-27T11:07:25.557106Z","shell.execute_reply":"2022-07-27T11:07:25.617013Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_test.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-27T11:07:42.218184Z","iopub.execute_input":"2022-07-27T11:07:42.218756Z","iopub.status.idle":"2022-07-27T11:07:42.284129Z","shell.execute_reply.started":"2022-07-27T11:07:42.218705Z","shell.execute_reply":"2022-07-27T11:07:42.282904Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Prediction with ML models**","metadata":{}},{"cell_type":"code","source":"k_fold = KFold(n_splits = 15, random_state = 11, shuffle = True)","metadata":{"execution":{"iopub.status.busy":"2022-07-27T11:08:27.500992Z","iopub.execute_input":"2022-07-27T11:08:27.501461Z","iopub.status.idle":"2022-07-27T11:08:27.507179Z","shell.execute_reply.started":"2022-07-27T11:08:27.501421Z","shell.execute_reply":"2022-07-27T11:08:27.505993Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def cv_rmse(model, X = x_train):\n    rmse = np.sqrt(-cross_val_score(model, x_train, y_train, scoring = \"neg_mean_squared_error\", cv = k_fold))\n    return rmse\n\n\ndef rmsle(y, y_pred):\n    return np.sqrt(mean_squared_log_error(y, y_pred, squared = False))","metadata":{"execution":{"iopub.status.busy":"2022-07-27T11:08:38.224766Z","iopub.execute_input":"2022-07-27T11:08:38.225234Z","iopub.status.idle":"2022-07-27T11:08:38.232203Z","shell.execute_reply.started":"2022-07-27T11:08:38.225195Z","shell.execute_reply":"2022-07-27T11:08:38.231150Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"xgb = make_pipeline(RobustScaler(),\n                    XGBRegressor(colsample_bytree = 0.5, n_estimators = 6000,\n                                 max_depth = 4, learning_rate = 0.01, gamma = 0.45,\n                                 subsample = 0.5, random_state = 11, reg_alpha = 0.00006,\n                                 reg_lambda = None, nthread = -1))","metadata":{"execution":{"iopub.status.busy":"2022-07-27T11:08:52.140275Z","iopub.execute_input":"2022-07-27T11:08:52.140656Z","iopub.status.idle":"2022-07-27T11:08:52.146918Z","shell.execute_reply.started":"2022-07-27T11:08:52.140624Z","shell.execute_reply":"2022-07-27T11:08:52.145960Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **XGBoost**","metadata":{}},{"cell_type":"code","source":"# get CV score of the xgb model\nscore = cv_rmse(xgb)\nprint(\"Xgboost model's cross validation score: \", score.mean())","metadata":{"execution":{"iopub.status.busy":"2022-07-27T11:09:05.651770Z","iopub.execute_input":"2022-07-27T11:09:05.652204Z","iopub.status.idle":"2022-07-27T11:19:30.035908Z","shell.execute_reply.started":"2022-07-27T11:09:05.652169Z","shell.execute_reply":"2022-07-27T11:19:30.034704Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# LightGbm","metadata":{}},{"cell_type":"code","source":"lgbm = make_pipeline(RobustScaler(),\n                     LGBMRegressor(num_leaves = 6, bagging_fraction = 0.7,\n                                   bagging_freq = 4, min_sum_hessian_in_leaf = 11,\n                                   learning_rate = 0.01, n_estimators = 7500, max_bin = 200,\n                                   random_state = 11))","metadata":{"execution":{"iopub.status.busy":"2022-07-27T11:24:26.895701Z","iopub.execute_input":"2022-07-27T11:24:26.896695Z","iopub.status.idle":"2022-07-27T11:24:26.903005Z","shell.execute_reply.started":"2022-07-27T11:24:26.896654Z","shell.execute_reply":"2022-07-27T11:24:26.901608Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# get CV score of the lgbm model\nscore = cv_rmse(lgbm)\nprint(\"Light GBM model's cross validation score: \", score.mean())","metadata":{"execution":{"iopub.status.busy":"2022-07-27T11:24:38.622998Z","iopub.execute_input":"2022-07-27T11:24:38.624171Z","iopub.status.idle":"2022-07-27T11:25:41.690085Z","shell.execute_reply.started":"2022-07-27T11:24:38.624128Z","shell.execute_reply":"2022-07-27T11:25:41.688421Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ridge = make_pipeline(RobustScaler(),\n                      RidgeCV(alphas = [1e-10, 1e-8, 1e-5, 1e-2, 9e-4,\n                                                        5e-4, 3e-4, 1e-4, 1e-3, 1e-2, 0.1,\n                                                        0.3, 0.6, 1, 3, 5, 7, 14, 18, 25, 30, \n                                                        45, 50, 70, 90], cv = k_fold))","metadata":{"execution":{"iopub.status.busy":"2022-07-27T11:25:51.817175Z","iopub.execute_input":"2022-07-27T11:25:51.817604Z","iopub.status.idle":"2022-07-27T11:25:51.824776Z","shell.execute_reply.started":"2022-07-27T11:25:51.817569Z","shell.execute_reply":"2022-07-27T11:25:51.823261Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# get CV score of the ridge model\nscore = cv_rmse(ridge)\nprint(\"Ridge model's cross validation score: \", score.mean())","metadata":{"execution":{"iopub.status.busy":"2022-07-27T11:25:57.864082Z","iopub.execute_input":"2022-07-27T11:25:57.864516Z","iopub.status.idle":"2022-07-27T11:27:35.373766Z","shell.execute_reply.started":"2022-07-27T11:25:57.864479Z","shell.execute_reply":"2022-07-27T11:27:35.372406Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"lasso = make_pipeline(RobustScaler(),\n                      LassoCV(alphas = [1e-10, 1e-8, 1e-5, 1e-2, 9e-4,\n                                                        5e-4, 3e-4, 1e-4, 1e-3, 1e-2, 0.1,\n                                                        0.3, 0.6, 1, 3, 5, 7, 14, 18, 25, 30,\n                                                        45, 50, 70, 90], n_jobs = -1, cv = k_fold))","metadata":{"execution":{"iopub.status.busy":"2022-07-27T11:28:24.301006Z","iopub.execute_input":"2022-07-27T11:28:24.301671Z","iopub.status.idle":"2022-07-27T11:28:24.310578Z","shell.execute_reply.started":"2022-07-27T11:28:24.301622Z","shell.execute_reply":"2022-07-27T11:28:24.309463Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# get CV score of the lasso model\nscore = cv_rmse(lasso)\nprint(\"Lasso model's cross validation score: \", score.mean())","metadata":{"execution":{"iopub.status.busy":"2022-07-27T11:28:27.186658Z","iopub.execute_input":"2022-07-27T11:28:27.187407Z","iopub.status.idle":"2022-07-27T11:28:41.250660Z","shell.execute_reply.started":"2022-07-27T11:28:27.187360Z","shell.execute_reply":"2022-07-27T11:28:41.247568Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gbr = make_pipeline(RobustScaler(),\n                    GradientBoostingRegressor(n_estimators = 7000, learning_rate = 0.01,\n                                              max_depth = 5, min_samples_split = 12, min_samples_leaf = 16,\n                                              loss = \"huber\", max_features = \"sqrt\", random_state = 11))","metadata":{"execution":{"iopub.status.busy":"2022-07-27T11:29:27.576379Z","iopub.execute_input":"2022-07-27T11:29:27.576815Z","iopub.status.idle":"2022-07-27T11:29:27.583530Z","shell.execute_reply.started":"2022-07-27T11:29:27.576775Z","shell.execute_reply":"2022-07-27T11:29:27.582373Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# get CV score of the gbr model\nscore = cv_rmse(gbr)\nprint(\"Gradient boosting model's cross validation score: \", score.mean())","metadata":{"execution":{"iopub.status.busy":"2022-07-27T11:29:30.273176Z","iopub.execute_input":"2022-07-27T11:29:30.273716Z","iopub.status.idle":"2022-07-27T11:40:23.289171Z","shell.execute_reply.started":"2022-07-27T11:29:30.273666Z","shell.execute_reply":"2022-07-27T11:40:23.287847Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rf = make_pipeline(RobustScaler(),\n                   RandomForestRegressor(n_estimators = 2500, max_depth = 15,\n                                         min_samples_split = 6, min_samples_leaf = 6,\n                                         random_state = 11))","metadata":{"execution":{"iopub.status.busy":"2022-07-27T11:41:51.467783Z","iopub.execute_input":"2022-07-27T11:41:51.468190Z","iopub.status.idle":"2022-07-27T11:41:51.475099Z","shell.execute_reply.started":"2022-07-27T11:41:51.468159Z","shell.execute_reply":"2022-07-27T11:41:51.473804Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# get CV score of the rf model\nscore = cv_rmse(rf)\nprint(\"Random forest model's cross validation score: \", score.mean())","metadata":{"execution":{"iopub.status.busy":"2022-07-27T11:41:58.621986Z","iopub.execute_input":"2022-07-27T11:41:58.622398Z","iopub.status.idle":"2022-07-27T11:49:52.925422Z","shell.execute_reply.started":"2022-07-27T11:41:58.622364Z","shell.execute_reply":"2022-07-27T11:49:52.924441Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"svr = make_pipeline(RobustScaler(), SVR(C = 30, gamma = 0.0002, epsilon = 0.009))","metadata":{"execution":{"iopub.status.busy":"2022-07-27T11:50:31.725036Z","iopub.execute_input":"2022-07-27T11:50:31.725437Z","iopub.status.idle":"2022-07-27T11:50:31.731788Z","shell.execute_reply.started":"2022-07-27T11:50:31.725405Z","shell.execute_reply":"2022-07-27T11:50:31.730328Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# get CV score of the svr model\nscore = cv_rmse(svr)\nprint(\"Support vector machines model's cross validation score: \", score.mean())","metadata":{"execution":{"iopub.status.busy":"2022-07-27T11:50:41.011767Z","iopub.execute_input":"2022-07-27T11:50:41.012193Z","iopub.status.idle":"2022-07-27T11:50:46.009965Z","shell.execute_reply.started":"2022-07-27T11:50:41.012158Z","shell.execute_reply":"2022-07-27T11:50:46.008936Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"stacked = StackingCVRegressor(regressors = (xgb, lgbm, ridge, svr, lasso, gbr, rf),\n                              meta_regressor = xgb, use_features_in_secondary = True)","metadata":{"execution":{"iopub.status.busy":"2022-07-27T11:50:49.092016Z","iopub.execute_input":"2022-07-27T11:50:49.093070Z","iopub.status.idle":"2022-07-27T11:50:49.099223Z","shell.execute_reply.started":"2022-07-27T11:50:49.093023Z","shell.execute_reply":"2022-07-27T11:50:49.098237Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gbr_model = gbr.fit(x_train, y_train)\n\n#RMSLE score of the gbr model on full train data\ngbr_score = rmsle(y_train, gbr_model.predict(x_train))\nprint(\"RMSLE score of xgboost model on full data:\", gbr_score)","metadata":{"execution":{"iopub.status.busy":"2022-07-27T11:50:52.403066Z","iopub.execute_input":"2022-07-27T11:50:52.403482Z","iopub.status.idle":"2022-07-27T11:51:37.306736Z","shell.execute_reply.started":"2022-07-27T11:50:52.403446Z","shell.execute_reply":"2022-07-27T11:51:37.305551Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"svr_model = svr.fit(x_train, y_train)\n\n#RMSLE score of the svr model on full train data\nsvr_score = rmsle(y_train, svr_model.predict(x_train))\nprint(\"RMSLE score of svr model on full data:\", svr_score)","metadata":{"execution":{"iopub.status.busy":"2022-07-27T11:52:13.414596Z","iopub.execute_input":"2022-07-27T11:52:13.414989Z","iopub.status.idle":"2022-07-27T11:52:14.247801Z","shell.execute_reply.started":"2022-07-27T11:52:13.414956Z","shell.execute_reply":"2022-07-27T11:52:14.246608Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"xgb_model = xgb.fit(x_train, y_train)\n\n#RMSLE score of the xgb model on full train data\nxgb_score = rmsle(y_train, xgb_model.predict(x_train))\nprint(\"RMSLE score of xgboost model on full data:\", xgb_score)","metadata":{"execution":{"iopub.status.busy":"2022-07-27T11:54:52.344623Z","iopub.execute_input":"2022-07-27T11:54:52.345721Z","iopub.status.idle":"2022-07-27T11:55:35.204614Z","shell.execute_reply.started":"2022-07-27T11:54:52.345669Z","shell.execute_reply":"2022-07-27T11:55:35.203100Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"lgbm_model = lgbm.fit(x_train, y_train)\n\n#RMSLE score of the lgbm model on full train data\nlgbm_score = rmsle(y_train, lgbm_model.predict(x_train))\nprint(\"RMSLE score of lgbm model on full data:\", lgbm_score)","metadata":{"execution":{"iopub.status.busy":"2022-07-27T11:55:50.447214Z","iopub.execute_input":"2022-07-27T11:55:50.448158Z","iopub.status.idle":"2022-07-27T11:55:55.682070Z","shell.execute_reply.started":"2022-07-27T11:55:50.448103Z","shell.execute_reply":"2022-07-27T11:55:55.680646Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ridge_model = ridge.fit(x_train, y_train)\n\n#RMSLE score of the ridge model on full train data\nridge_score = rmsle(y_train, ridge_model.predict(x_train))\nprint(\"RMSLE score of ridge model on full data:\", ridge_score)","metadata":{"execution":{"iopub.status.busy":"2022-07-27T11:55:58.420352Z","iopub.execute_input":"2022-07-27T11:55:58.421523Z","iopub.status.idle":"2022-07-27T11:56:06.412312Z","shell.execute_reply.started":"2022-07-27T11:55:58.421476Z","shell.execute_reply":"2022-07-27T11:56:06.410901Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"lasso_model = lasso.fit(x_train, y_train)\n\n#RMSLE score of the lasso model on full train data\nlasso_score = rmsle(y_train, lasso_model.predict(x_train))\nprint(\"RMSLE score of lasso model on full data:\", lasso_score)","metadata":{"execution":{"iopub.status.busy":"2022-07-27T11:56:10.871571Z","iopub.execute_input":"2022-07-27T11:56:10.872024Z","iopub.status.idle":"2022-07-27T11:56:12.410593Z","shell.execute_reply.started":"2022-07-27T11:56:10.871987Z","shell.execute_reply":"2022-07-27T11:56:12.409046Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rf_model = rf.fit(x_train, y_train)\n\n#RMSLE score of the rf model on full train data\nrf_score = rmsle(y_train, rf_model.predict(x_train))\nprint(\"RMSLE score of random forest model on full data:\", rf_score)","metadata":{"execution":{"iopub.status.busy":"2022-07-27T11:56:14.710029Z","iopub.execute_input":"2022-07-27T11:56:14.710424Z","iopub.status.idle":"2022-07-27T11:56:49.451339Z","shell.execute_reply.started":"2022-07-27T11:56:14.710392Z","shell.execute_reply":"2022-07-27T11:56:49.449949Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"stacked_model = stacked.fit(np.array(x_train), np.array(y_train))\n\n#RMSLE score of the stacked model on full train data\nstacked_score = rmsle(y_train, stacked_model.predict(x_train))\nprint(\"RMSLE score of stacked models on full data:\", stacked_score)","metadata":{"execution":{"iopub.status.busy":"2022-07-27T11:56:53.805036Z","iopub.execute_input":"2022-07-27T11:56:53.806008Z","iopub.status.idle":"2022-07-27T12:10:19.602597Z","shell.execute_reply.started":"2022-07-27T11:56:53.805960Z","shell.execute_reply":"2022-07-27T12:10:19.601447Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred = np.floor(np.expm1(stacked_model.predict(x_test)))\ny_pred[0:5]","metadata":{"execution":{"iopub.status.busy":"2022-07-27T12:11:16.308147Z","iopub.execute_input":"2022-07-27T12:11:16.309072Z","iopub.status.idle":"2022-07-27T12:11:18.414636Z","shell.execute_reply.started":"2022-07-27T12:11:16.309014Z","shell.execute_reply":"2022-07-27T12:11:18.413187Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = pd.DataFrame()\nsubmission[\"Id\"] = test_df[\"Id\"]\nsubmission[\"SalePrice\"] = y_pred\nsubmission.to_csv(\"submission.csv\", index = False)\nsubmission.head(n = 10)","metadata":{"execution":{"iopub.status.busy":"2022-07-27T12:11:37.027111Z","iopub.execute_input":"2022-07-27T12:11:37.027553Z","iopub.status.idle":"2022-07-27T12:11:37.054573Z","shell.execute_reply.started":"2022-07-27T12:11:37.027518Z","shell.execute_reply":"2022-07-27T12:11:37.053211Z"},"trusted":true},"execution_count":null,"outputs":[]}]}