{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# # **1. Importing**","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\n\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom scipy.stats import norm\nfrom sklearn.preprocessing import StandardScaler\nfrom scipy import stats\nimport warnings\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))","metadata":{"execution":{"iopub.status.busy":"2022-07-13T03:31:43.066906Z","iopub.execute_input":"2022-07-13T03:31:43.067351Z","iopub.status.idle":"2022-07-13T03:31:43.075570Z","shell.execute_reply.started":"2022-07-13T03:31:43.067299Z","shell.execute_reply":"2022-07-13T03:31:43.074723Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#train.csv - the training set\n#test.csv - the test set\ntrain = pd.read_csv(\"../input/house-prices-advanced-regression-techniques/train.csv\")\ntest = pd.read_csv(\"../input/house-prices-advanced-regression-techniques/test.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-07-13T03:31:43.102203Z","iopub.execute_input":"2022-07-13T03:31:43.102827Z","iopub.status.idle":"2022-07-13T03:31:43.143994Z","shell.execute_reply.started":"2022-07-13T03:31:43.102793Z","shell.execute_reply":"2022-07-13T03:31:43.142697Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# # **2. View on the datasets**","metadata":{}},{"cell_type":"code","source":"train.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-13T03:31:43.155056Z","iopub.execute_input":"2022-07-13T03:31:43.155421Z","iopub.status.idle":"2022-07-13T03:31:43.179800Z","shell.execute_reply.started":"2022-07-13T03:31:43.155390Z","shell.execute_reply":"2022-07-13T03:31:43.178858Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-13T03:31:43.205480Z","iopub.execute_input":"2022-07-13T03:31:43.205892Z","iopub.status.idle":"2022-07-13T03:31:43.234150Z","shell.execute_reply.started":"2022-07-13T03:31:43.205859Z","shell.execute_reply":"2022-07-13T03:31:43.232922Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Let us run .info() function to see more information about our datasets","metadata":{}},{"cell_type":"code","source":"train.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-13T03:31:43.242264Z","iopub.execute_input":"2022-07-13T03:31:43.243080Z","iopub.status.idle":"2022-07-13T03:31:43.266356Z","shell.execute_reply.started":"2022-07-13T03:31:43.243037Z","shell.execute_reply":"2022-07-13T03:31:43.265219Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-13T03:31:43.281215Z","iopub.execute_input":"2022-07-13T03:31:43.281788Z","iopub.status.idle":"2022-07-13T03:31:43.300504Z","shell.execute_reply.started":"2022-07-13T03:31:43.281755Z","shell.execute_reply":"2022-07-13T03:31:43.299534Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Missing values on train dataset","metadata":{}},{"cell_type":"code","source":"train.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2022-07-13T03:31:43.318796Z","iopub.execute_input":"2022-07-13T03:31:43.319167Z","iopub.status.idle":"2022-07-13T03:31:43.333304Z","shell.execute_reply.started":"2022-07-13T03:31:43.319136Z","shell.execute_reply":"2022-07-13T03:31:43.332026Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Let us check the size of our datasets","metadata":{}},{"cell_type":"code","source":"print(\"The size of train data (row,column) is:\" + str(train.shape))\nprint(\"The size of test data (row, column) is:\" + str(test.shape))","metadata":{"execution":{"iopub.status.busy":"2022-07-13T03:31:43.361048Z","iopub.execute_input":"2022-07-13T03:31:43.361485Z","iopub.status.idle":"2022-07-13T03:31:43.367028Z","shell.execute_reply.started":"2022-07-13T03:31:43.361445Z","shell.execute_reply":"2022-07-13T03:31:43.365773Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Our datasets show that we have 1460 houses to train the model and 1459 houses to predict the price.","metadata":{}},{"cell_type":"code","source":"#check the decoration\ntrain.columns","metadata":{"execution":{"iopub.status.busy":"2022-07-13T03:31:43.396806Z","iopub.execute_input":"2022-07-13T03:31:43.397590Z","iopub.status.idle":"2022-07-13T03:31:43.405488Z","shell.execute_reply.started":"2022-07-13T03:31:43.397546Z","shell.execute_reply":"2022-07-13T03:31:43.404562Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Let us performe basic statistics on the train data ","metadata":{}},{"cell_type":"code","source":"train.describe().T","metadata":{"execution":{"iopub.status.busy":"2022-07-13T03:31:43.441224Z","iopub.execute_input":"2022-07-13T03:31:43.441971Z","iopub.status.idle":"2022-07-13T03:31:43.545230Z","shell.execute_reply.started":"2022-07-13T03:31:43.441922Z","shell.execute_reply":"2022-07-13T03:31:43.544111Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['SalePrice'].describe()","metadata":{"execution":{"iopub.status.busy":"2022-07-13T03:31:43.547484Z","iopub.execute_input":"2022-07-13T03:31:43.548164Z","iopub.status.idle":"2022-07-13T03:31:43.560445Z","shell.execute_reply.started":"2022-07-13T03:31:43.548117Z","shell.execute_reply":"2022-07-13T03:31:43.559201Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#histogram\nsns.histplot(train['SalePrice']);","metadata":{"execution":{"iopub.status.busy":"2022-07-13T03:31:43.561687Z","iopub.execute_input":"2022-07-13T03:31:43.562186Z","iopub.status.idle":"2022-07-13T03:31:43.871609Z","shell.execute_reply.started":"2022-07-13T03:31:43.562148Z","shell.execute_reply":"2022-07-13T03:31:43.870542Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.dtypes.value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-07-13T03:31:43.874054Z","iopub.execute_input":"2022-07-13T03:31:43.874378Z","iopub.status.idle":"2022-07-13T03:31:43.882990Z","shell.execute_reply.started":"2022-07-13T03:31:43.874350Z","shell.execute_reply":"2022-07-13T03:31:43.881741Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We have 43 features of type object, 38 numerical values (intigers and floats).","metadata":{}},{"cell_type":"markdown","source":"# # **3. Data cleaning**","metadata":{}},{"cell_type":"markdown","source":"From the description above, we have seen that we have some messing values, we need to replace or delete to avoid any problem during our caluculations.  ","metadata":{}},{"cell_type":"markdown","source":"**Missing data**","metadata":{}},{"cell_type":"code","source":"total = train \\\n    .isnull() \\\n    .sum() \\\n    .sort_values(ascending=False)\n\npercent = (train.isnull().sum()/train.isnull().count()).sort_values(ascending=False)\n\nmissing_data = pd.concat([total, percent], axis=1, keys=['Total', 'Percent'])\n\nmissing_data.head(20)","metadata":{"execution":{"iopub.status.busy":"2022-07-13T03:31:43.884513Z","iopub.execute_input":"2022-07-13T03:31:43.885087Z","iopub.status.idle":"2022-07-13T03:31:43.915629Z","shell.execute_reply.started":"2022-07-13T03:31:43.885056Z","shell.execute_reply":"2022-07-13T03:31:43.914703Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = train.drop((missing_data[missing_data['Total'] > 0]).index,1)\n\n#train = train.drop(train.loc[train['Electrical'].isnull()].index)\n\ntrain \\\n    .isnull() \\\n    .sum() \\\n    .max()","metadata":{"execution":{"iopub.status.busy":"2022-07-13T03:31:43.919065Z","iopub.execute_input":"2022-07-13T03:31:43.919539Z","iopub.status.idle":"2022-07-13T03:31:43.934183Z","shell.execute_reply.started":"2022-07-13T03:31:43.919501Z","shell.execute_reply":"2022-07-13T03:31:43.933313Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# # **4. Core work**","metadata":{}},{"cell_type":"markdown","source":"By performing Correlation matrix, we identify and visualize patterns in our data","metadata":{}},{"cell_type":"code","source":"#correlation matrix (heatmap)\ncorrmat = train.corr()\nf, ax = plt.subplots(figsize=(12, 9))\nsns.heatmap(corrmat, vmax=.8, square=True);","metadata":{"execution":{"iopub.status.busy":"2022-07-13T03:31:43.935829Z","iopub.execute_input":"2022-07-13T03:31:43.936189Z","iopub.status.idle":"2022-07-13T03:31:44.769095Z","shell.execute_reply.started":"2022-07-13T03:31:43.936157Z","shell.execute_reply":"2022-07-13T03:31:44.768294Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"At first sight, there are two red colored squares that get my attention. The first one refers to the 'TotalBsmtSF' and '1stFlrSF' variables, and the second one refers to the 'GarageX' variables. Both cases show how significant the correlation is between these variables. Actually, this correlation is so strong that it can indicate a situation of multicollinearity. If we think about these variables, we can conclude that they give almost the same information so multicollinearity really occurs. Heatmaps are great to detect this kind of situations and in problems dominated by feature selection, like ours, they are an essential tool.\n\nAnother thing that got my attention was the 'SalePrice' correlations. We can see our well-known 'GrLivArea', 'TotalBsmtSF', and 'OverallQual' saying a big 'Hi!', but we can also see many other variables that should be taken into account. That's what we will do next.","metadata":{}},{"cell_type":"code","source":"#saleprice correlation matrix\n# Let k be number of variables for heatmap\nk = 10 \ncols = corrmat.nlargest(k, 'SalePrice')['SalePrice'].index\ncm = np.corrcoef(train[cols].values.T)\nsns.set(font_scale=1.25)\nhm = sns.heatmap(cm, cbar=True, annot=True, square=True, fmt='.2f', annot_kws={'size': 10}, yticklabels=cols.values, xticklabels=cols.values)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-13T03:31:44.770372Z","iopub.execute_input":"2022-07-13T03:31:44.771671Z","iopub.status.idle":"2022-07-13T03:31:45.412144Z","shell.execute_reply.started":"2022-07-13T03:31:44.771567Z","shell.execute_reply":"2022-07-13T03:31:45.410999Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"According to our crystal ball, these are the variables most correlated with 'SalePrice'.","metadata":{}},{"cell_type":"markdown","source":"Scatter plots between 'SalePrice' and correlated variables.","metadata":{}},{"cell_type":"code","source":"#scatterplot\nsns.set()\ncols = ['SalePrice', 'OverallQual', 'GrLivArea', 'GarageCars', 'TotalBsmtSF', 'FullBath', 'YearBuilt']\nsns.pairplot(train[cols], height = 2.5)\nplt.show();","metadata":{"execution":{"iopub.status.busy":"2022-07-13T03:31:45.416446Z","iopub.execute_input":"2022-07-13T03:31:45.416944Z","iopub.status.idle":"2022-07-13T03:31:53.544725Z","shell.execute_reply.started":"2022-07-13T03:31:45.416897Z","shell.execute_reply":"2022-07-13T03:31:53.543611Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The scatter plots give us a reasonable idea about variables relationships.","metadata":{}},{"cell_type":"markdown","source":"Normal probability plot - Data distribution should closely follow the diagonal that represents the normal distribution.","metadata":{}},{"cell_type":"code","source":"#histogram and normal probability plot\nsns.distplot(train['SalePrice'], fit=norm);\nfig = plt.figure()\nres = stats.probplot(train['SalePrice'], plot=plt)","metadata":{"execution":{"iopub.status.busy":"2022-07-13T03:31:53.545838Z","iopub.execute_input":"2022-07-13T03:31:53.546131Z","iopub.status.idle":"2022-07-13T03:31:54.071020Z","shell.execute_reply.started":"2022-07-13T03:31:53.546104Z","shell.execute_reply":"2022-07-13T03:31:54.069812Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"'SalePrice' is not normal. It shows 'peakedness', positive skewness and does not follow the diagonal line.","metadata":{}},{"cell_type":"code","source":"#applying log transformation\ntrain['SalePrice'] = np.log(train['SalePrice'])","metadata":{"execution":{"iopub.status.busy":"2022-07-13T03:31:54.072248Z","iopub.execute_input":"2022-07-13T03:31:54.072574Z","iopub.status.idle":"2022-07-13T03:31:54.078290Z","shell.execute_reply.started":"2022-07-13T03:31:54.072537Z","shell.execute_reply":"2022-07-13T03:31:54.077533Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#transformed histogram and normal probability plot\nsns.distplot(train['SalePrice'], fit=norm);\nfig = plt.figure()\nres = stats.probplot(train['SalePrice'], plot=plt)","metadata":{"execution":{"iopub.status.busy":"2022-07-13T03:31:54.079277Z","iopub.execute_input":"2022-07-13T03:31:54.080271Z","iopub.status.idle":"2022-07-13T03:31:54.609864Z","shell.execute_reply.started":"2022-07-13T03:31:54.080239Z","shell.execute_reply":"2022-07-13T03:31:54.608833Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#convert categorical variable into dummy\ntrain = pd.get_dummies(train)","metadata":{"execution":{"iopub.status.busy":"2022-07-13T03:31:54.611062Z","iopub.execute_input":"2022-07-13T03:31:54.611414Z","iopub.status.idle":"2022-07-13T03:31:54.642868Z","shell.execute_reply.started":"2022-07-13T03:31:54.611382Z","shell.execute_reply":"2022-07-13T03:31:54.641754Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.to_csv(\"submission.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-13T03:34:23.158805Z","iopub.execute_input":"2022-07-13T03:34:23.159570Z","iopub.status.idle":"2022-07-13T03:34:23.215040Z","shell.execute_reply.started":"2022-07-13T03:34:23.159530Z","shell.execute_reply":"2022-07-13T03:34:23.214165Z"},"trusted":true},"execution_count":null,"outputs":[]}]}