{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"<div style=\"border-radius:20px;\n            border : black solid;\n            background-color: ##FFFFFF;\n            font-size:200%;\n            text-align: left\">\n\n<h1 style='; border:0; border-radius: 15px; text-shadow: 1px 1px black; font-weight: bold; color:green'><center> HOUSE PRICES PREDICTION </center></h1>","metadata":{}},{"cell_type":"markdown","source":"![](https://miro.medium.com/max/1024/1*Juv1bpp5--0Fl8cA4EmTPw.jpeg)","metadata":{}},{"cell_type":"markdown","source":"# **Index**\n\n- [Importing necessary libraries](#import)\n- [Importing the data](#data)\n- [Raw Data Visualization](#vis)\n    - [Scatterplot of raw data](#scatt)\n    - [Null values heat-map](#heat)\n    - [Comparison of Null values between training and testing data](#barnull)\n- [Data Pre-processing](#preprocess)\n- [Splitting data into x (Values) and y (labels)](#split)\n- [Creating Model](#model)\n    - [Gradient Boosting Regression](#gbr)\n    - [Random Forest](#rf)\n- [Submission](#submit)","metadata":{}},{"cell_type":"markdown","source":"<div style=\"border-radius:10px;\n            border : black solid;\n            background-color:  #FFA07A;\n            font-size:110%;\n            text-align: left\">\n​\n<h4 style='; border:0; border-radius: 10px; font-weight: bold; color:black'><center> Importing necessary libraries</center></h4><a id=\"import\"></a>\n","metadata":{}},{"cell_type":"code","source":"import tensorflow as tf\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport sklearn","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-13T12:44:17.872963Z","iopub.execute_input":"2022-07-13T12:44:17.873713Z","iopub.status.idle":"2022-07-13T12:44:17.879565Z","shell.execute_reply.started":"2022-07-13T12:44:17.873671Z","shell.execute_reply":"2022-07-13T12:44:17.878472Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<div style=\"border-radius:10px;\n            border : black solid;\n            background-color: #3CB371;\n            font-size:110%;\n            text-align: left\">\n​\n<h4 style='; border:0; border-radius: 10px; font-weight: bold; color:black'><center> Importing the data</center></h4><a id=\"data\"></a>\n","metadata":{}},{"cell_type":"code","source":"import os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))","metadata":{"execution":{"iopub.status.busy":"2022-07-13T12:44:17.881317Z","iopub.execute_input":"2022-07-13T12:44:17.881844Z","iopub.status.idle":"2022-07-13T12:44:17.893181Z","shell.execute_reply.started":"2022-07-13T12:44:17.881812Z","shell.execute_reply":"2022-07-13T12:44:17.892242Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataTrain = pd.read_csv('/kaggle/input/house-prices-advanced-regression-techniques/train.csv')\ndataTrain.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-13T12:44:17.894660Z","iopub.execute_input":"2022-07-13T12:44:17.895394Z","iopub.status.idle":"2022-07-13T12:44:17.941143Z","shell.execute_reply.started":"2022-07-13T12:44:17.895357Z","shell.execute_reply":"2022-07-13T12:44:17.940028Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataTest = pd.read_csv('/kaggle/input/house-prices-advanced-regression-techniques/test.csv')\ndataTest.tail()","metadata":{"execution":{"iopub.status.busy":"2022-07-13T12:44:17.942739Z","iopub.execute_input":"2022-07-13T12:44:17.943257Z","iopub.status.idle":"2022-07-13T12:44:17.989383Z","shell.execute_reply.started":"2022-07-13T12:44:17.943224Z","shell.execute_reply":"2022-07-13T12:44:17.988234Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<div style=\"border-radius:10px;\n            border : black solid;\n            background-color: #C71585;\n            font-size:110%;\n            text-align: left\">\n​\n<h4 style='; border:0; border-radius: 10px; font-weight: bold; color:black'><center>  Data Visualization</center></h4><a id=\"vis\"></a>\n","metadata":{}},{"cell_type":"code","source":"fig_ = dataTrain.hist(figsize=(25, 30), bins=50, color=\"darkcyan\",\n                         edgecolor=\"black\", xlabelsize=8, ylabelsize=8)","metadata":{"execution":{"iopub.status.busy":"2022-07-13T12:44:17.991468Z","iopub.execute_input":"2022-07-13T12:44:17.993000Z","iopub.status.idle":"2022-07-13T12:44:27.794115Z","shell.execute_reply.started":"2022-07-13T12:44:17.992960Z","shell.execute_reply":"2022-07-13T12:44:27.793218Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig_ = dataTest.hist(figsize=(25, 30), bins=50, color=\"red\",\n                         edgecolor=\"black\", xlabelsize=8, ylabelsize=8)","metadata":{"execution":{"iopub.status.busy":"2022-07-13T12:44:27.795334Z","iopub.execute_input":"2022-07-13T12:44:27.796248Z","iopub.status.idle":"2022-07-13T12:44:37.327039Z","shell.execute_reply.started":"2022-07-13T12:44:27.796204Z","shell.execute_reply":"2022-07-13T12:44:37.325882Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### ***Scatterplot of raw data*** <a id=\"scatt\"></a>","metadata":{}},{"cell_type":"code","source":"sns.set(rc={'figure.figsize':(11.7,8.27)})\nsns.scatterplot(x=dataTrain.Id, y=dataTrain.SalePrice, size=dataTrain.SalePrice, hue=dataTrain.OverallCond, style=dataTrain.YrSold, sizes=(60,300), palette=\"magma\")\nplt.title('Scatterplot of raw data')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-13T12:44:37.328604Z","iopub.execute_input":"2022-07-13T12:44:37.328959Z","iopub.status.idle":"2022-07-13T12:44:38.011099Z","shell.execute_reply.started":"2022-07-13T12:44:37.328927Z","shell.execute_reply":"2022-07-13T12:44:38.009736Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### ***Null values heat-map*** <a id=\"heat\"></a>","metadata":{}},{"cell_type":"code","source":"fig, axes = plt.subplots(1, 2, sharex=True, figsize=(20,10))\nsns.heatmap(ax=axes[0], yticklabels=False, data=dataTrain.isnull(), cbar=False, cmap=\"viridis\")\nsns.heatmap(ax=axes[1], yticklabels=False, data=dataTest.isnull(), cbar=False, cmap=\"tab20c\")\naxes[0].set_title('Heatmap of missing values in training data')\naxes[1].set_title('Heatmap of missing values in testing data')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-13T12:44:38.012842Z","iopub.execute_input":"2022-07-13T12:44:38.013491Z","iopub.status.idle":"2022-07-13T12:44:39.423863Z","shell.execute_reply.started":"2022-07-13T12:44:38.013443Z","shell.execute_reply":"2022-07-13T12:44:39.422580Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### ***Comparison of Null values between training and testing data*** <a id=\"barnull\"></a>","metadata":{}},{"cell_type":"code","source":"def show_values(axs, orient=\"v\", space=.01):\n    def _single(ax):\n        if orient == \"v\":\n            for p in ax.patches:\n                _x = p.get_x() + p.get_width() / 2\n                _y = p.get_y() + p.get_height() + (p.get_height()*0.01)\n                value = '{:.1f}'.format(p.get_height())\n                ax.text(_x, _y, value, ha=\"center\") \n        elif orient == \"h\":\n            for p in ax.patches:\n                _x = p.get_x() + p.get_width() + float(space)\n                _y = p.get_y() + p.get_height() - (p.get_height()*0.5)\n                value = '{:.1f}'.format(p.get_width())\n                ax.text(_x, _y, value, ha=\"left\")\n\n    if isinstance(axs, np.ndarray):\n        for idx, ax in np.ndenumerate(axs):\n            _single(ax)\n    else:\n        _single(axs)","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-07-13T12:44:39.425581Z","iopub.execute_input":"2022-07-13T12:44:39.425919Z","iopub.status.idle":"2022-07-13T12:44:39.437672Z","shell.execute_reply.started":"2022-07-13T12:44:39.425889Z","shell.execute_reply":"2022-07-13T12:44:39.436394Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, axes = plt.subplots(1, 2, sharex=True, figsize=(20,10))\nnanTrain = {}\nfor column in dataTrain.columns[1:]:\n    perc =  dataTrain[column].isna().sum()/len(dataTrain[column])\n    if perc >= 0.01:\n        nanTrain[str(column)] = perc\nnanTrain = {key: value*100 for key, value in sorted(nanTrain.items(), key=lambda item: item[1], reverse=True)}\na = sns.barplot(ax=axes[0], y=list(nanTrain.keys()), x=list(nanTrain.values()), palette=\"coolwarm\", ci=None)\nplt.xlabel(\"NaN Values (%)\")\nplt.ylabel(\"Labels\")\nplt.title('NaN values in training data:')\n#===================================================================================================\nnanTest = {}\nfor column in dataTest.columns[1:]:\n    perc =  dataTest[column].isna().sum()/len(dataTest[column])\n    if perc >= 0.01:\n        nanTest[str(column)] = perc\nnanTest = {key: value*100 for key, value in sorted(nanTest.items(), key=lambda item: item[1], reverse=True)}\nb = sns.barplot(ax=axes[1], y=list(nanTest.keys()), x=list(nanTest.values()), palette=\"flare\", ci=None)\n\naxes[0].set_title('Missing data in training set')\naxes[1].set_title('Missing data in training set')\naxes[0].set_xlabel('NaN Values (%)')\naxes[0].set_ylabel('Labels')\naxes[1].set_xlabel('NaN Values (%)')\naxes[1].set_ylabel('Labels')\n\nshow_values(a, \"h\", space=0.3)\nshow_values(b, \"h\", space=0.3)\n\nplt.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-07-13T12:44:39.442317Z","iopub.execute_input":"2022-07-13T12:44:39.443518Z","iopub.status.idle":"2022-07-13T12:44:40.278967Z","shell.execute_reply.started":"2022-07-13T12:44:39.443471Z","shell.execute_reply":"2022-07-13T12:44:40.277431Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<div style=\"border-radius:10px;\n            border : black solid;\n            background-color:  \t#BDB76B;\n            font-size:110%;\n            text-align: left\">\n​\n<h2 style='; border:0; border-radius: 10px; font-weight: bold; color:black'><center> Data Pre-processing</center></h2><a id=\"preprocess\"></a>\n","metadata":{}},{"cell_type":"markdown","source":"#### Checking Correlation between labels and the target label.","metadata":{}},{"cell_type":"code","source":"THRESHOLD = 0.5\n\ndata = dataTrain.corr()[\"SalePrice\"].sort_values(ascending=False)\nindices = data.index\nlabels = []\ncorr = []\nfor i in range(1, len(indices)):\n    if data[indices[i]]>THRESHOLD:\n        labels.append(indices[i])\n        corr.append(data[i])\nsns.barplot(x=corr, y=labels)\nplt.title('Lables with correlation coefficient > Threshold (0.5)')","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-07-13T12:44:40.280730Z","iopub.execute_input":"2022-07-13T12:44:40.281841Z","iopub.status.idle":"2022-07-13T12:44:40.607276Z","shell.execute_reply.started":"2022-07-13T12:44:40.281785Z","shell.execute_reply":"2022-07-13T12:44:40.605959Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Dropping the columns that have insignificant correlation with out target variable (unnecessary columns).","metadata":{}},{"cell_type":"code","source":"unnecessary = []\nlab = dataTrain.SalePrice\nidCol = dataTest.Id\ndataTrain = dataTrain.drop(columns=[str(item) for item in dataTrain.columns[1:] if str(item) not in labels])\ndataTest = dataTest.drop(columns=[str(item) for item in dataTest.columns[1:] if str(item) not in labels])\ndataTrain = dataTrain.drop(columns=['Id'])\ndataTest = dataTest.drop(columns=['Id'])","metadata":{"execution":{"iopub.status.busy":"2022-07-13T12:44:40.608784Z","iopub.execute_input":"2022-07-13T12:44:40.609100Z","iopub.status.idle":"2022-07-13T12:44:40.621412Z","shell.execute_reply.started":"2022-07-13T12:44:40.609073Z","shell.execute_reply":"2022-07-13T12:44:40.620197Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataTrain = dataTrain.fillna(method='bfill')\ndataTest = dataTest.fillna(method='bfill')","metadata":{"execution":{"iopub.status.busy":"2022-07-13T12:44:40.622962Z","iopub.execute_input":"2022-07-13T12:44:40.623359Z","iopub.status.idle":"2022-07-13T12:44:40.634953Z","shell.execute_reply.started":"2022-07-13T12:44:40.623325Z","shell.execute_reply":"2022-07-13T12:44:40.633733Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Checking if further NaN values persist.","metadata":{}},{"cell_type":"code","source":"sum(dataTrain.isnull().sum()), sum(dataTest.isnull().sum())","metadata":{"execution":{"iopub.status.busy":"2022-07-13T12:44:40.636456Z","iopub.execute_input":"2022-07-13T12:44:40.637402Z","iopub.status.idle":"2022-07-13T12:44:40.651105Z","shell.execute_reply.started":"2022-07-13T12:44:40.637354Z","shell.execute_reply":"2022-07-13T12:44:40.649741Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<div style=\"border-radius:10px;\n            border : black solid;\n            background-color:  #FFA500;\n            font-size:110%;\n            text-align: left\">\n​\n<h2 style='; border:0; border-radius: 10px; font-weight: bold; color:black'><center> Splitting data into x (Values) and y (labels)</center></h2><a id=\"split\"></a>\n\n","metadata":{}},{"cell_type":"code","source":"yTrain = lab\nxTest = dataTest.to_numpy()\nxTrain = dataTrain.to_numpy()\n\nxTrain.shape, yTrain.shape, xTest.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-13T12:44:40.652980Z","iopub.execute_input":"2022-07-13T12:44:40.653726Z","iopub.status.idle":"2022-07-13T12:44:40.663493Z","shell.execute_reply.started":"2022-07-13T12:44:40.653678Z","shell.execute_reply":"2022-07-13T12:44:40.662629Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<div style=\"border-radius:10px;\n            border : black solid;\n            background-color: #DDA0DD;\n            font-size:110%;\n            text-align: left\">\n​\n<h2 style='; border:0; border-radius: 10px; font-weight: bold; color:black'><center> Creating Model</center></h2><a id=\"model\"></a>\n\n","metadata":{}},{"cell_type":"markdown","source":"**Gradient Boosting Regression**<a id=\"gbr\"></a>\n","metadata":{}},{"cell_type":"markdown","source":"Gradient boosting is one of the most popular machine learning algorithms for tabular datasets. It is powerful enough to find any nonlinear relationship between your model target and features and has great usability that can deal with missing values, outliers, and high cardinality categorical values on your features without any special treatment. While you can build barebone gradient boosting trees using some popular libraries such as XGBoost or LightGBM without knowing any details of the algorithm, you still want to know how it works when you start tuning hyper-parameters, customizing the loss functions, etc., to get better quality on your model.\n![](https://miro.medium.com/max/640/1*NLI9QFoWDltdXJf3_rwbbw.png)","metadata":{}},{"cell_type":"code","source":"from sklearn.ensemble import GradientBoostingRegressor\nfrom sklearn.model_selection import KFold\nfrom sklearn.metrics import mean_squared_error, r2_score","metadata":{"execution":{"iopub.status.busy":"2022-07-13T12:44:40.664539Z","iopub.execute_input":"2022-07-13T12:44:40.665362Z","iopub.status.idle":"2022-07-13T12:44:40.672735Z","shell.execute_reply.started":"2022-07-13T12:44:40.665312Z","shell.execute_reply":"2022-07-13T12:44:40.671501Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"reg = GradientBoostingRegressor(random_state=42, loss='ls', learning_rate=0.1)","metadata":{"execution":{"iopub.status.busy":"2022-07-13T12:44:40.674041Z","iopub.execute_input":"2022-07-13T12:44:40.674641Z","iopub.status.idle":"2022-07-13T12:44:40.685960Z","shell.execute_reply.started":"2022-07-13T12:44:40.674605Z","shell.execute_reply":"2022-07-13T12:44:40.685005Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"reg.fit(xTrain, yTrain)","metadata":{"execution":{"iopub.status.busy":"2022-07-13T12:44:40.687290Z","iopub.execute_input":"2022-07-13T12:44:40.687945Z","iopub.status.idle":"2022-07-13T12:44:40.964292Z","shell.execute_reply.started":"2022-07-13T12:44:40.687905Z","shell.execute_reply":"2022-07-13T12:44:40.963066Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred = reg.predict(xTest)","metadata":{"execution":{"iopub.status.busy":"2022-07-13T12:44:40.965725Z","iopub.execute_input":"2022-07-13T12:44:40.966109Z","iopub.status.idle":"2022-07-13T12:44:40.974386Z","shell.execute_reply.started":"2022-07-13T12:44:40.966075Z","shell.execute_reply":"2022-07-13T12:44:40.973232Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"f=reg.score(xTrain, yTrain)\nf","metadata":{"execution":{"iopub.status.busy":"2022-07-13T12:44:40.975696Z","iopub.execute_input":"2022-07-13T12:44:40.976098Z","iopub.status.idle":"2022-07-13T12:44:40.990665Z","shell.execute_reply.started":"2022-07-13T12:44:40.976064Z","shell.execute_reply":"2022-07-13T12:44:40.989426Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Random Forest**<a id=\"rf\"></a>\n","metadata":{}},{"cell_type":"markdown","source":"Random forest is a commonly-used machine learning algorithm trademarked by Leo Breiman and Adele Cutler, which combines the output of multiple decision trees to reach a single result. Its ease of use and flexibility have fueled its adoption, as it handles both classification and regression problems.\n![](https://www.analyticssteps.com/backend/media/thumbnail/2050098/3299256_1589184813_Random%20Forest...jpg)","metadata":{}},{"cell_type":"code","source":"from sklearn.ensemble import RandomForestClassifier\nmodel = RandomForestClassifier(max_depth=15)\nmodel.fit(xTrain,yTrain)\npreds = model.predict(xTrain)\nprint('R2 Score: ', sklearn.metrics.r2_score(yTrain,preds)) ","metadata":{"execution":{"iopub.status.busy":"2022-07-13T12:44:40.992251Z","iopub.execute_input":"2022-07-13T12:44:40.993699Z","iopub.status.idle":"2022-07-13T12:44:44.199163Z","shell.execute_reply.started":"2022-07-13T12:44:40.993655Z","shell.execute_reply":"2022-07-13T12:44:44.197993Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<div style=\"border-radius:10px;\n            border : black solid;\n            background-color: \t#C0C0C0;\n            font-size:110%;\n            text-align: left\">\n​\n<h2 style='; border:0; border-radius: 10px; font-weight: bold; color:black'><center> Submission</center></h2><a id=\"submit\"></a>\n","metadata":{}},{"cell_type":"code","source":"Final = pd.DataFrame(columns=[\"Id\",\"SalePrice\"])\nFinal[\"Id\"] = idCol\nFinal[\"SalePrice\"] = model.predict(xTest)\nFinal[\"Id\"] = Final[\"Id\"].astype(int)\nFinal","metadata":{"execution":{"iopub.status.busy":"2022-07-13T12:44:44.200910Z","iopub.execute_input":"2022-07-13T12:44:44.201325Z","iopub.status.idle":"2022-07-13T12:44:44.723115Z","shell.execute_reply.started":"2022-07-13T12:44:44.201288Z","shell.execute_reply":"2022-07-13T12:44:44.721786Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Final.to_csv('submission.csv', index=False)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-13T12:44:44.724752Z","iopub.execute_input":"2022-07-13T12:44:44.725736Z","iopub.status.idle":"2022-07-13T12:44:44.737446Z","shell.execute_reply.started":"2022-07-13T12:44:44.725694Z","shell.execute_reply":"2022-07-13T12:44:44.736282Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<div style=\"border-radius:10px;\n            border : black solid;\n            background-color: #FDF5E6;\n            font-size:110%;\n            text-align: left\">\n​\n<h2 style='; border:0; border-radius: 10px; font-weight: bold; color:black'><center> Gradient Boosting Regressor : 93% AND Random forest : 99%</center></h2>","metadata":{}},{"cell_type":"markdown","source":"<div style=\"border-radius:10px;\n            border : black solid;\n            background-color: #DA70D6;\n            font-size:200%;\n            text-align: left\">\n\n<h1 style='; border:0; border-radius: 10px; text-shadow: 1px 1px black; font-weight: bold; color:black'><center> YOUR FEEDBACKS ARE HIGHLY APPRECIATED </center></h1>","metadata":{}},{"cell_type":"markdown","source":"![](https://st2.depositphotos.com/1006899/7664/i/600/depositphotos_76643019-stock-photo-thank-you-words.jpg)","metadata":{}}]}