{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport seaborn as sns\nfrom pylab import rcParams\nimport matplotlib.pyplot as plt\nimport matplotlib.animation as animation\nfrom matplotlib import rc\nimport unittest\n\n%matplotlib inline\n\nsns.set(style='whitegrid', palette='muted', font_scale=1.5)\n\n\nRANDOM_SEED = 42\n\nnp.random.seed(RANDOM_SEED)\n\ndef run_tests():\n  unittest.main(argv=[''], verbosity=1, exit=False)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-06T21:10:13.081644Z","iopub.execute_input":"2022-07-06T21:10:13.083018Z","iopub.status.idle":"2022-07-06T21:10:14.459536Z","shell.execute_reply.started":"2022-07-06T21:10:13.082907Z","shell.execute_reply":"2022-07-06T21:10:14.457909Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = pd.read_csv(\"../input/house-prices-advanced-regression-techniques/train.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-07-06T21:10:14.462119Z","iopub.execute_input":"2022-07-06T21:10:14.462614Z","iopub.status.idle":"2022-07-06T21:10:14.507229Z","shell.execute_reply.started":"2022-07-06T21:10:14.462567Z","shell.execute_reply":"2022-07-06T21:10:14.505772Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-06T21:10:14.508768Z","iopub.execute_input":"2022-07-06T21:10:14.509420Z","iopub.status.idle":"2022-07-06T21:10:14.550081Z","shell.execute_reply.started":"2022-07-06T21:10:14.509383Z","shell.execute_reply":"2022-07-06T21:10:14.549030Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Exploring the Dataset","metadata":{}},{"cell_type":"code","source":"#explore the sale price column\ndf_train['SalePrice'].describe()","metadata":{"execution":{"iopub.status.busy":"2022-07-06T21:10:14.552582Z","iopub.execute_input":"2022-07-06T21:10:14.553355Z","iopub.status.idle":"2022-07-06T21:10:14.572457Z","shell.execute_reply.started":"2022-07-06T21:10:14.553316Z","shell.execute_reply":"2022-07-06T21:10:14.571151Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#show the data distribution for SalePrice feature\nsns.distplot(df_train['SalePrice']);","metadata":{"execution":{"iopub.status.busy":"2022-07-06T21:10:14.576481Z","iopub.execute_input":"2022-07-06T21:10:14.577293Z","iopub.status.idle":"2022-07-06T21:10:14.979232Z","shell.execute_reply.started":"2022-07-06T21:10:14.577239Z","shell.execute_reply":"2022-07-06T21:10:14.977850Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"from the above data distribution we can see that the denisity lies between 100K and 250K","metadata":{}},{"cell_type":"code","source":"#Let's now measure the correaltion between the dataset features\n#and the SalePrice\ncorrmat = df_train.corr()\nf, ax = plt.subplots(figsize=(12, 9))\nsns.heatmap(corrmat, vmax=.8, square=True);","metadata":{"execution":{"iopub.status.busy":"2022-07-06T21:10:14.980996Z","iopub.execute_input":"2022-07-06T21:10:14.981439Z","iopub.status.idle":"2022-07-06T21:10:15.600961Z","shell.execute_reply.started":"2022-07-06T21:10:14.981406Z","shell.execute_reply":"2022-07-06T21:10:15.599994Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#now let's get the most correlated variables with the sale price\nk = 9 #number of variables for heatmap\ncols = corrmat.nlargest(k, 'SalePrice')['SalePrice'].index\nf, ax = plt.subplots(figsize=(14, 10))\nsns.heatmap(df_train[cols].corr(), vmax=.8, square=True);","metadata":{"execution":{"iopub.status.busy":"2022-07-06T21:10:15.602376Z","iopub.execute_input":"2022-07-06T21:10:15.603334Z","iopub.status.idle":"2022-07-06T21:10:16.068576Z","shell.execute_reply.started":"2022-07-06T21:10:15.603296Z","shell.execute_reply":"2022-07-06T21:10:16.067107Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"var = 'GrLivArea'\ndata = pd.concat([df_train['SalePrice'], df_train[var]], axis=1)\ndata.plot.scatter(x=var, y='SalePrice', ylim=(0,800000), s=32);\n\nvar = 'OverallQual'\ndata = pd.concat([df_train['SalePrice'], df_train[var]], axis=1)\ndata.plot.scatter(x=var, y='SalePrice', ylim=(0,800000), s=32);\n","metadata":{"execution":{"iopub.status.busy":"2022-07-06T21:10:16.070598Z","iopub.execute_input":"2022-07-06T21:10:16.072237Z","iopub.status.idle":"2022-07-06T21:10:16.602388Z","shell.execute_reply.started":"2022-07-06T21:10:16.072176Z","shell.execute_reply":"2022-07-06T21:10:16.601034Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#let's look for missing data\ntotal = df_train.isnull().sum().sort_values(ascending=False)\npercent = (df_train.isnull().sum()/df_train.isnull().count()).sort_values(ascending=False)\nmissing_data = pd.concat([total, percent], axis=1, keys=['Total', 'Percent'])\nmissing_data.head(20)","metadata":{"execution":{"iopub.status.busy":"2022-07-06T21:10:16.604041Z","iopub.execute_input":"2022-07-06T21:10:16.604564Z","iopub.status.idle":"2022-07-06T21:10:16.645589Z","shell.execute_reply.started":"2022-07-06T21:10:16.604527Z","shell.execute_reply":"2022-07-06T21:10:16.644681Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Feature Scaling\nlet's now make sure that all our features; i mean here the Xs are on a similar scale by using the feature scaling techniques\nan this will be by making each feature into aproxmiltly -1<= x <= 1\nfor example  x1(size) = (0-2000) we can standarized this by dividing x1/2000\nX2(# of bedrooms) = (1-5) we can standarized this by dividing x2/5\n\nanother technique is the mean normalization:\nx1 - mean / STD or Range\n\nthe objective her is to get all your features ( x1 - x2 - x3 - x4....etc)\nin this or a proximate to this range -0.5<= Xi <= 0.5","metadata":{}},{"cell_type":"code","source":"# let's do a feature scaling by using Mean Normalization trick\n\n#assign the feature to a variable x\nx = df_train['GrLivArea']\ny = df_train['SalePrice']\n#get the normalized mean\nx = (x - x.mean()) / x.std()\n# use \"numpy.c_\" for indexing the data points and have 2-D array instead of \n# 1-D array\n#x = np.c_[np.ones(x.shape[0]), x]\n\n#combine the 2 datasets\ndf1 = pd.concat([x, y], axis=1)\n#check of the data ha sbeen combined properly\ndf1.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-06T21:10:16.648311Z","iopub.execute_input":"2022-07-06T21:10:16.649460Z","iopub.status.idle":"2022-07-06T21:10:16.664199Z","shell.execute_reply.started":"2022-07-06T21:10:16.649407Z","shell.execute_reply":"2022-07-06T21:10:16.662278Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-06T21:10:16.666169Z","iopub.execute_input":"2022-07-06T21:10:16.667309Z","iopub.status.idle":"2022-07-06T21:10:16.674534Z","shell.execute_reply.started":"2022-07-06T21:10:16.667268Z","shell.execute_reply":"2022-07-06T21:10:16.673451Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Predicting the Sales Price by using Linear regression technique\nusing linear regression or multivariate linear regression as you saw from the \nEDA we did above that there is a relationship between some variables and the SalPrice variable\n\nwe will start by building our sample linear regression on greater living area feature","metadata":{}},{"cell_type":"code","source":"#using the Ordinary Least Squares techniques\n#find mean of x(GrLivArea) and the mean of y ( SalePrice)\nxmean = np.mean(x)\nymean = np.mean(y)\n\n# Calculate the terms needed for the numator and denominator of beta\ndf1['xycov'] = (x - xmean) * (y - ymean)\ndf1['xvar'] = (x- xmean)**2\n\n# Calculate beta and alpha\nbeta = df1['xycov'].sum() / df1['xvar'].sum()\nalpha = ymean - (beta * xmean)\nprint(f'alpha = {alpha}')\nprint(f'beta = {beta}')","metadata":{"execution":{"iopub.status.busy":"2022-07-06T21:10:16.676199Z","iopub.execute_input":"2022-07-06T21:10:16.677251Z","iopub.status.idle":"2022-07-06T21:10:16.690400Z","shell.execute_reply.started":"2022-07-06T21:10:16.677212Z","shell.execute_reply":"2022-07-06T21:10:16.689507Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df1.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-06T21:10:16.691686Z","iopub.execute_input":"2022-07-06T21:10:16.692671Z","iopub.status.idle":"2022-07-06T21:10:16.710997Z","shell.execute_reply.started":"2022-07-06T21:10:16.692636Z","shell.execute_reply":"2022-07-06T21:10:16.709470Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Great, we now have an estimate for alpha and beta! Our model can be written as Yₑ = 2.003 + 0.323 X,​ \n#and we can make predictions:\ndf1['ypred'] = alpha + beta * x\ndf1['Variance'] = df1['SalePrice'] - df1['ypred']\n\ndf1.head(20)","metadata":{"execution":{"iopub.status.busy":"2022-07-06T21:10:16.712696Z","iopub.execute_input":"2022-07-06T21:10:16.713587Z","iopub.status.idle":"2022-07-06T21:10:16.735483Z","shell.execute_reply.started":"2022-07-06T21:10:16.713545Z","shell.execute_reply":"2022-07-06T21:10:16.734139Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Let’s plot our prediction ypred against the actual values of y,\n#to get a better visual understanding of our model.\n\n# Plot regression against actual data\nplt.figure(figsize=(12, 6))\nplt.plot(x, df1['ypred'])     # regression line\nplt.plot(x, y, 'ro')   # scatter plot showing actual data\nplt.title('Actual vs Predicted')\nplt.xlabel('GrLivArea')\nplt.ylabel('SalePrice')\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-06T21:13:52.390757Z","iopub.execute_input":"2022-07-06T21:13:52.391945Z","iopub.status.idle":"2022-07-06T21:13:52.696882Z","shell.execute_reply.started":"2022-07-06T21:13:52.391881Z","shell.execute_reply":"2022-07-06T21:13:52.695656Z"},"trusted":true},"execution_count":null,"outputs":[]}]}