{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport matplotlib.pyplot as plt\nimport seaborn as sb\nimport warnings\nwarnings.filterwarnings('ignore')\nfrom matplotlib import rcParams\n\n# figure size in inches\nrcParams['figure.figsize'] = 11.7,8.27\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","execution":{"iopub.status.busy":"2022-07-24T17:38:45.914957Z","iopub.execute_input":"2022-07-24T17:38:45.915370Z","iopub.status.idle":"2022-07-24T17:38:45.924531Z","shell.execute_reply.started":"2022-07-24T17:38:45.915334Z","shell.execute_reply":"2022-07-24T17:38:45.923659Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Read the data description ( description of each feature)","metadata":{}},{"cell_type":"markdown","source":"#### Read train, test dataset","metadata":{}},{"cell_type":"code","source":"train_df= pd.read_csv('../input/house-prices-advanced-regression-techniques/train.csv')\ntest_df= pd.read_csv('../input/house-prices-advanced-regression-techniques/test.csv')\n","metadata":{"execution":{"iopub.status.busy":"2022-07-24T17:38:45.926012Z","iopub.execute_input":"2022-07-24T17:38:45.927182Z","iopub.status.idle":"2022-07-24T17:38:45.974617Z","shell.execute_reply.started":"2022-07-24T17:38:45.927138Z","shell.execute_reply":"2022-07-24T17:38:45.973443Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-24T17:38:45.975983Z","iopub.execute_input":"2022-07-24T17:38:45.976296Z","iopub.status.idle":"2022-07-24T17:38:45.982324Z","shell.execute_reply.started":"2022-07-24T17:38:45.976268Z","shell.execute_reply":"2022-07-24T17:38:45.981587Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-24T17:38:45.983337Z","iopub.execute_input":"2022-07-24T17:38:45.984042Z","iopub.status.idle":"2022-07-24T17:38:46.012581Z","shell.execute_reply.started":"2022-07-24T17:38:45.984009Z","shell.execute_reply":"2022-07-24T17:38:46.011731Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-24T17:38:46.014681Z","iopub.execute_input":"2022-07-24T17:38:46.015161Z","iopub.status.idle":"2022-07-24T17:38:46.042457Z","shell.execute_reply.started":"2022-07-24T17:38:46.015129Z","shell.execute_reply":"2022-07-24T17:38:46.041416Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"### studying about train & test dataset in general\ntrain_df.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-24T17:38:46.043585Z","iopub.execute_input":"2022-07-24T17:38:46.043877Z","iopub.status.idle":"2022-07-24T17:38:46.065066Z","shell.execute_reply.started":"2022-07-24T17:38:46.043851Z","shell.execute_reply":"2022-07-24T17:38:46.063917Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"##### observations\n\n1. Train Dataset has 80 features including target\n\n2. There are 1460 train values , out of some features had missing values ","metadata":{}},{"cell_type":"code","source":"test_df.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-24T17:38:46.066246Z","iopub.execute_input":"2022-07-24T17:38:46.067210Z","iopub.status.idle":"2022-07-24T17:38:46.088815Z","shell.execute_reply.started":"2022-07-24T17:38:46.067166Z","shell.execute_reply":"2022-07-24T17:38:46.087933Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### observations \n\n1. test dataset has 79 features \n\n2. some values are null","metadata":{}},{"cell_type":"code","source":"#### describe the train dataset\ntrain_df.describe()","metadata":{"execution":{"iopub.status.busy":"2022-07-24T17:38:46.089969Z","iopub.execute_input":"2022-07-24T17:38:46.090448Z","iopub.status.idle":"2022-07-24T17:38:46.187932Z","shell.execute_reply.started":"2022-07-24T17:38:46.090417Z","shell.execute_reply":"2022-07-24T17:38:46.186935Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#### describe test dataset\ntest_df.describe()","metadata":{"execution":{"iopub.status.busy":"2022-07-24T17:38:46.191104Z","iopub.execute_input":"2022-07-24T17:38:46.191731Z","iopub.status.idle":"2022-07-24T17:38:46.282964Z","shell.execute_reply.started":"2022-07-24T17:38:46.191693Z","shell.execute_reply":"2022-07-24T17:38:46.281841Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#### calculating the number of null values in train dataset\ntrain_df.isnull().sum()[train_df.isnull().sum()!=0]*100/(train_df.shape[0])","metadata":{"execution":{"iopub.status.busy":"2022-07-24T17:38:46.286169Z","iopub.execute_input":"2022-07-24T17:38:46.286515Z","iopub.status.idle":"2022-07-24T17:38:46.311298Z","shell.execute_reply.started":"2022-07-24T17:38:46.286473Z","shell.execute_reply":"2022-07-24T17:38:46.308543Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.isnull().sum()[train_df.isnull().sum()!=0]","metadata":{"execution":{"iopub.status.busy":"2022-07-24T17:38:46.312609Z","iopub.execute_input":"2022-07-24T17:38:46.313068Z","iopub.status.idle":"2022-07-24T17:38:46.330002Z","shell.execute_reply.started":"2022-07-24T17:38:46.313039Z","shell.execute_reply":"2022-07-24T17:38:46.328930Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### observations:\n\n1.Taking the features which has missing values, it could be seen that Features like LotFrontage,Alley,FireplaceQu,PoolQC,Fence,MiscFeature has missing values >10%\n\n2. From the list of features having missing values >10%, i could see that features among which are categorical are having NAN as one of their category [ got details from data_description.txt]\n\n\na. Electrical,LotFrontage,MasVnrArea,GarageYrBlt is not categorical, may be needed to remove the features nan values\n","metadata":{}},{"cell_type":"code","source":"#### here let's drop the Lot FRontage from the train_df as it has high missing values [ >10% and no of missing values=259], as imputation is difficult to achieve.\ntrain_df.drop('LotFrontage',axis=1,inplace=True)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-24T17:38:46.331120Z","iopub.execute_input":"2022-07-24T17:38:46.331668Z","iopub.status.idle":"2022-07-24T17:38:46.338077Z","shell.execute_reply.started":"2022-07-24T17:38:46.331636Z","shell.execute_reply":"2022-07-24T17:38:46.337034Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-24T17:38:46.339631Z","iopub.execute_input":"2022-07-24T17:38:46.340262Z","iopub.status.idle":"2022-07-24T17:38:46.368290Z","shell.execute_reply.started":"2022-07-24T17:38:46.340220Z","shell.execute_reply":"2022-07-24T17:38:46.367557Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#### removed null values from Electrical and MasVnrArea columns\ntrain_df= train_df[~(train_df['Electrical'].isnull() | train_df['MasVnrArea'].isnull())]\ntrain_df.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-24T17:38:46.369482Z","iopub.execute_input":"2022-07-24T17:38:46.369956Z","iopub.status.idle":"2022-07-24T17:38:46.378279Z","shell.execute_reply.started":"2022-07-24T17:38:46.369926Z","shell.execute_reply":"2022-07-24T17:38:46.377461Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.isnull().sum()[train_df.isnull().sum()!=0]","metadata":{"execution":{"iopub.status.busy":"2022-07-24T17:38:46.379482Z","iopub.execute_input":"2022-07-24T17:38:46.380373Z","iopub.status.idle":"2022-07-24T17:38:46.402470Z","shell.execute_reply.started":"2022-07-24T17:38:46.380340Z","shell.execute_reply":"2022-07-24T17:38:46.401635Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"### handling null values\ntrain_df['Alley'].fillna('No Alley Access',inplace=True)\nfor c in ['BsmtQual','BsmtCond','BsmtExposure','BsmtFinType1','BsmtFinType2']:\n    train_df[c].fillna('No Basement',inplace=True)\n    \n\ntrain_df['FireplaceQu'].fillna('No fireplace',inplace=True)\nfor c1 in ['GarageType','GarageYrBlt','GarageFinish','GarageQual','GarageCond']:\n    train_df[c1].fillna('No Garage',inplace=True)\n    \n\ntrain_df['PoolQC'].fillna('No Pool',inplace=True)\ntrain_df['Fence'].fillna('No Fence',inplace=True)\ntrain_df['MiscFeature'].fillna('No Misc Feature',inplace=True)\ntrain_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-24T17:38:46.403923Z","iopub.execute_input":"2022-07-24T17:38:46.404435Z","iopub.status.idle":"2022-07-24T17:38:46.436593Z","shell.execute_reply.started":"2022-07-24T17:38:46.404406Z","shell.execute_reply":"2022-07-24T17:38:46.435750Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#drop dupliactes if available\ntrain_df.drop_duplicates(inplace=True)\ntrain_df.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-24T17:38:46.437917Z","iopub.execute_input":"2022-07-24T17:38:46.438419Z","iopub.status.idle":"2022-07-24T17:38:46.463577Z","shell.execute_reply.started":"2022-07-24T17:38:46.438389Z","shell.execute_reply":"2022-07-24T17:38:46.462547Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Exploratory Data Analysis [EDA]","metadata":{}},{"cell_type":"code","source":"train_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-24T17:38:46.464917Z","iopub.execute_input":"2022-07-24T17:38:46.465191Z","iopub.status.idle":"2022-07-24T17:38:46.487440Z","shell.execute_reply.started":"2022-07-24T17:38:46.465166Z","shell.execute_reply":"2022-07-24T17:38:46.486281Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-24T17:38:46.488928Z","iopub.execute_input":"2022-07-24T17:38:46.489798Z","iopub.status.idle":"2022-07-24T17:38:46.514649Z","shell.execute_reply.started":"2022-07-24T17:38:46.489751Z","shell.execute_reply":"2022-07-24T17:38:46.513573Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for c in ['MSSubClass','OverallQual','OverallCond','BsmtFullBath','BsmtHalfBath','FullBath','HalfBath','BedroomAbvGr','TotRmsAbvGrd']:\n    train_df[c]= train_df[c].astype(object)","metadata":{"execution":{"iopub.status.busy":"2022-07-24T17:38:46.516335Z","iopub.execute_input":"2022-07-24T17:38:46.517365Z","iopub.status.idle":"2022-07-24T17:38:46.528779Z","shell.execute_reply.started":"2022-07-24T17:38:46.517321Z","shell.execute_reply":"2022-07-24T17:38:46.527587Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df[['MSSubClass','OverallQual','OverallCond','BsmtFullBath','BsmtHalfBath','FullBath','HalfBath','BedroomAbvGr','TotRmsAbvGrd']].info()","metadata":{"execution":{"iopub.status.busy":"2022-07-24T17:38:46.534835Z","iopub.execute_input":"2022-07-24T17:38:46.536148Z","iopub.status.idle":"2022-07-24T17:38:46.550557Z","shell.execute_reply.started":"2022-07-24T17:38:46.536108Z","shell.execute_reply":"2022-07-24T17:38:46.549491Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"### lets study the distribution of categorical columns\n\n#### add the categorical columns as list\ncategorical_cols= [col for col in train_df.columns if train_df[col].dtype==object]\ncategorical_cols","metadata":{"execution":{"iopub.status.busy":"2022-07-24T17:38:46.551592Z","iopub.execute_input":"2022-07-24T17:38:46.552655Z","iopub.status.idle":"2022-07-24T17:38:46.565145Z","shell.execute_reply.started":"2022-07-24T17:38:46.552617Z","shell.execute_reply":"2022-07-24T17:38:46.563999Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#### create a count plot for the following categories\nfor i in categorical_cols:\n    plt.figure(figsize=(12,8))\n    plt.title(f'Countplot for {i}')\n    sb.countplot(data=train_df,x=i)\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-24T17:38:46.566886Z","iopub.execute_input":"2022-07-24T17:38:46.567196Z","iopub.status.idle":"2022-07-24T17:38:58.135148Z","shell.execute_reply.started":"2022-07-24T17:38:46.567169Z","shell.execute_reply":"2022-07-24T17:38:58.133763Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Observations:\n\n1. Almost all categories are un equally distributed in general","metadata":{}},{"cell_type":"code","source":"#### computing sale price of house (average) for each category\nfor c in categorical_cols:\n    tempdf= train_df.groupby(c).aggregate({'SalePrice':'median'})\n    tempdf.plot(kind='barh',figsize=(12,8),title=f'Mean Sale price for each category under {c}')\n    ","metadata":{"execution":{"iopub.status.busy":"2022-07-24T17:38:58.136615Z","iopub.execute_input":"2022-07-24T17:38:58.136950Z","iopub.status.idle":"2022-07-24T17:39:09.191267Z","shell.execute_reply.started":"2022-07-24T17:38:58.136919Z","shell.execute_reply":"2022-07-24T17:39:09.189938Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Observations:\n\n1.  Among MSSubClass categories, the Class 60 (2-STORY 1946 & NEWER) has highest median price\n\n\n2.  Among MSZoning class, FV (Floating Village Residential has highest median sale price)\n\n\n3.  Among the sreet category, street with paved road has highesh median sale price\n\n\n4.  Among alley category, paved alley has more sale price than no alley access \n\n\n5.  Among Lotshape, irregular plot cost more than regular (why so)\n\n\n6.  Among utlities, house with all public services available are more costlier than less utitlies available near to them.\n\n\n7.  Among LotConfig, CulDSac has most median sale price\n\n\n8.  AMong LandSlop, house with medium slope and severely sloped has higher sale price\n\n\n9.  Among condition 1, RRNn & PosA has highest median sale price\n\n\n10. Among condition 2, PosN, PosA has highest median sale price.\n\n\n11. Among BldgType, Twnhse has highest sale price.\n\n\n12. Among HouseStyle, 2 storied building and 2.5  fin has highest sale price.\n\n\n13. Coming to overallquality, house having excellent has highest sale price, while it is decreasing down when the overallquality decreases\n\n\n14. Based on overall condition, 9 th category has highest sale price and it is decreasing down as condition quality reduces, (but at 5 overall condition, it is increasing, need to study why ot occurs: keeping as special case)\n\n\n15. Based on roof design, shed roof design has higher sale price than flat roof design\n\n\n16. Based on material used for roof, it is seen that roof made of wood shingles have higher median price\n\n\n17. Based on exterior1st, stone & imstucc materials based house has high median sale price\n\n\n18. Based on masvnrtype, stone material based houses cost more in average\n\n\n19. Based on exterior material quality, excellent house cost more \n\n\n20.  Based on foundation material, poured concrete based foundation house cost more.\n\n\n21. Coming under heating type, gas furnance forced heating based houses cost more in average\n\n\n22. Houses having central air conditioing cost more in average\n\n\n","metadata":{}},{"cell_type":"code","source":"#### drawing the boxplot for each categorical feature against sale price\nfor c in categorical_cols:\n    plt.figure(figsize=(12,8))\n    sb.boxplot(data=train_df,x=c,y='SalePrice')\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-24T17:39:09.192966Z","iopub.execute_input":"2022-07-24T17:39:09.193373Z","iopub.status.idle":"2022-07-24T17:39:23.870858Z","shell.execute_reply.started":"2022-07-24T17:39:09.193339Z","shell.execute_reply":"2022-07-24T17:39:23.869874Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#### take numerical columns data \n\nnumerical_cols= [i for i in train_df.columns if i not in categorical_cols and i!='Id']\nnumerical_cols","metadata":{"execution":{"iopub.status.busy":"2022-07-24T17:39:23.872226Z","iopub.execute_input":"2022-07-24T17:39:23.872580Z","iopub.status.idle":"2022-07-24T17:39:23.879674Z","shell.execute_reply.started":"2022-07-24T17:39:23.872547Z","shell.execute_reply":"2022-07-24T17:39:23.878746Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"### draw heatmap for correlation \nplt.figure(figsize=(25,18))\nsb.heatmap(train_df[numerical_cols].corr(),annot=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-24T17:39:23.880925Z","iopub.execute_input":"2022-07-24T17:39:23.881327Z","iopub.status.idle":"2022-07-24T17:39:27.310666Z","shell.execute_reply.started":"2022-07-24T17:39:23.881234Z","shell.execute_reply":"2022-07-24T17:39:27.308740Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"##### Observations\n\n1. From the heatmap, it could be summarised as there is some correlation existing b/w independent features [exclude Saleprice].\n\n2. Also it could be summarised as correlation b/w some independent features & dependent feature [saleprice] is good.","metadata":{}},{"cell_type":"code","source":"##### Studying the Lot Area feature with respect to sale price\n\nplt.figure(figsize=(12,8))\nplt.title('Lot size with respect to Saleprice')\nplt.scatter(x=train_df['LotArea'],y=train_df['SalePrice'])\nplt.xlabel('LotArea')\nplt.ylabel('Saleprice')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-24T17:39:27.312161Z","iopub.execute_input":"2022-07-24T17:39:27.312723Z","iopub.status.idle":"2022-07-24T17:39:27.526355Z","shell.execute_reply.started":"2022-07-24T17:39:27.312683Z","shell.execute_reply":"2022-07-24T17:39:27.525138Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### observations\n\n1. Normally the lotarea should have higher influence (not major) on salesprice\n\n**2. Would need to see why the lotarea>50,000 sq feet price is lower than expected & saleprice> 5,00,000 USD is higher for lower lotarea under 50,000 sq.**\n\n","metadata":{}},{"cell_type":"code","source":"##### Studying the Lot Area feature with respect to MsDwelling\n\npd.DataFrame(train_df.groupby('MSSubClass').aggregate({'LotArea':'mean'})).plot(kind='barh',figsize=(12,8),title='Lot Area with dwell class')","metadata":{"execution":{"iopub.status.busy":"2022-07-24T17:39:27.527997Z","iopub.execute_input":"2022-07-24T17:39:27.528452Z","iopub.status.idle":"2022-07-24T17:39:27.797698Z","shell.execute_reply.started":"2022-07-24T17:39:27.528406Z","shell.execute_reply":"2022-07-24T17:39:27.796546Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"##### Studying the Lot Area feature with respect to LandContour\n\npd.DataFrame(train_df.groupby('LandContour').aggregate({'LotArea':'median'})).plot(kind='barh',figsize=(12,8),title='Lot Area with LandContour')","metadata":{"execution":{"iopub.status.busy":"2022-07-24T17:39:27.799327Z","iopub.execute_input":"2022-07-24T17:39:27.800170Z","iopub.status.idle":"2022-07-24T17:39:28.011535Z","shell.execute_reply.started":"2022-07-24T17:39:27.800131Z","shell.execute_reply":"2022-07-24T17:39:28.010363Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"##### ploting sale price with respect to yearbuilt\npd.DataFrame(train_df.groupby('YearBuilt').aggregate({'SalePrice':'median'})).plot(kind='barh',figsize=(12,30),title='YearBuilt with sale price')","metadata":{"execution":{"iopub.status.busy":"2022-07-24T17:39:28.013265Z","iopub.execute_input":"2022-07-24T17:39:28.014141Z","iopub.status.idle":"2022-07-24T17:39:29.295775Z","shell.execute_reply.started":"2022-07-24T17:39:28.014096Z","shell.execute_reply":"2022-07-24T17:39:29.294608Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"##### ploting sale price with respect to modified year \npd.DataFrame(train_df.groupby('YearRemodAdd').aggregate({'SalePrice':'median'})).plot(kind='barh',figsize=(12,30),title='YearModified with sale price')","metadata":{"execution":{"iopub.status.busy":"2022-07-24T17:39:29.297377Z","iopub.execute_input":"2022-07-24T17:39:29.297724Z","iopub.status.idle":"2022-07-24T17:39:30.104412Z","shell.execute_reply.started":"2022-07-24T17:39:29.297693Z","shell.execute_reply":"2022-07-24T17:39:30.103336Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(12,8))\nplt.title('Misc Val with respect to Saleprice')\nplt.scatter(x=train_df['MiscVal'],y=train_df['SalePrice'])\nplt.xlabel('MiscVal')\nplt.ylabel('Saleprice')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-24T17:39:30.106159Z","iopub.execute_input":"2022-07-24T17:39:30.106604Z","iopub.status.idle":"2022-07-24T17:39:30.338289Z","shell.execute_reply.started":"2022-07-24T17:39:30.106568Z","shell.execute_reply":"2022-07-24T17:39:30.336988Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(12,8))\nplt.title('MasVnrArea with respect to LotArea')\nplt.scatter(x=train_df['MasVnrArea'],y=train_df['LotArea'])\nplt.xlabel('MasVnrArea')\nplt.ylabel('LotArea')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-24T17:39:30.339863Z","iopub.execute_input":"2022-07-24T17:39:30.340298Z","iopub.status.idle":"2022-07-24T17:39:30.554655Z","shell.execute_reply.started":"2022-07-24T17:39:30.340261Z","shell.execute_reply":"2022-07-24T17:39:30.553576Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### LotArea and MasVnrArea are independent almost","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(12,8))\nplt.title('TotalBsmtSF with respect to LotArea')\nplt.scatter(x=train_df['TotalBsmtSF'],y=train_df['LotArea'])\nplt.xlabel('TotalBsmtSF')\nplt.ylabel('LotArea')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-24T17:39:30.555828Z","iopub.execute_input":"2022-07-24T17:39:30.556117Z","iopub.status.idle":"2022-07-24T17:39:30.757269Z","shell.execute_reply.started":"2022-07-24T17:39:30.556090Z","shell.execute_reply":"2022-07-24T17:39:30.755901Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"##### Observations\n\n1. TotalBasementArea and LotArea are almost independent ","metadata":{}},{"cell_type":"code","source":"##### studying the distribution of numerical data\nfor c in numerical_cols:\n    sb.displot(train_df[c],kind='kde')","metadata":{"execution":{"iopub.status.busy":"2022-07-24T17:39:30.758841Z","iopub.execute_input":"2022-07-24T17:39:30.759294Z","iopub.status.idle":"2022-07-24T17:39:37.442023Z","shell.execute_reply.started":"2022-07-24T17:39:30.759247Z","shell.execute_reply":"2022-07-24T17:39:37.440826Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"##### Observations\n\n1. LotArea distribution is left skewed\n\n2. YearBuilt distribution is bimodal\n\n3. YearRemodAdd is bimodal\n\n4. MasVnrArea is left skewed distribution\n\n5.  Basement Surface Area finished is bimodal\n\n6. Basement Unifinished Area is left skewed.\n\n7. TotalBasement Surface Area is left skewed.\n\n8. 2nd floor surface area is bimodal\n\n9. Garage Cars are multimodal\n\n10. GarageArea is normally distributed\n\n11. SalePrice has a left skewed distribution","metadata":{}},{"cell_type":"code","source":"#### drawing a boxenplot for each categorical data w.r.t saleprice\nfor c in categorical_cols:\n    sb.boxenplot(data=train_df,x=c,y='SalePrice')\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-24T17:39:37.443384Z","iopub.execute_input":"2022-07-24T17:39:37.443782Z","iopub.status.idle":"2022-07-24T17:39:52.595850Z","shell.execute_reply.started":"2022-07-24T17:39:37.443729Z","shell.execute_reply":"2022-07-24T17:39:52.594596Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#### plotting violin plot\nfor c in categorical_cols:\n    sb.violinplot(data=train_df,x=c,y='SalePrice')\n    plt.show()\n    ","metadata":{"execution":{"iopub.status.busy":"2022-07-24T17:39:52.597435Z","iopub.execute_input":"2022-07-24T17:39:52.597906Z","iopub.status.idle":"2022-07-24T17:40:11.184150Z","shell.execute_reply.started":"2022-07-24T17:39:52.597870Z","shell.execute_reply":"2022-07-24T17:40:11.182895Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#### plotting the boxplot for numerical data\nfor c in numerical_cols:\n    sb.boxplot(data=train_df,x=c)\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-24T17:40:11.185454Z","iopub.execute_input":"2022-07-24T17:40:11.185798Z","iopub.status.idle":"2022-07-24T17:40:14.206834Z","shell.execute_reply.started":"2022-07-24T17:40:11.185766Z","shell.execute_reply":"2022-07-24T17:40:14.205686Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"##### observations\n\n1. There are some outliers for the numerical features.","metadata":{}},{"cell_type":"code","source":"#### considering lotarea>50,000 sq ft\ntemp_df_lotarea= train_df[train_df['LotArea']>=50000]\ntemp_df_lotarea","metadata":{"execution":{"iopub.status.busy":"2022-07-24T17:40:14.208365Z","iopub.execute_input":"2022-07-24T17:40:14.210019Z","iopub.status.idle":"2022-07-24T17:40:14.240249Z","shell.execute_reply.started":"2022-07-24T17:40:14.209967Z","shell.execute_reply":"2022-07-24T17:40:14.239131Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#### comparing of the original train_df data statitics\ntrain_df[['LotArea','SalePrice']].describe()","metadata":{"execution":{"iopub.status.busy":"2022-07-24T17:40:14.241750Z","iopub.execute_input":"2022-07-24T17:40:14.242831Z","iopub.status.idle":"2022-07-24T17:40:14.264293Z","shell.execute_reply.started":"2022-07-24T17:40:14.242772Z","shell.execute_reply":"2022-07-24T17:40:14.263285Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"temp_df_lotarea[['LotArea','SalePrice']].describe()","metadata":{"execution":{"iopub.status.busy":"2022-07-24T17:40:14.267525Z","iopub.execute_input":"2022-07-24T17:40:14.267909Z","iopub.status.idle":"2022-07-24T17:40:14.287957Z","shell.execute_reply.started":"2022-07-24T17:40:14.267878Z","shell.execute_reply":"2022-07-24T17:40:14.286467Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### My question, in general lot area should be a dominating factor for house price, but why higher lot area has low price?\n\n#### Taking this later to study","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Feature Engineering & Feature Selection","metadata":{}},{"cell_type":"code","source":"##### creating new features out of existing features\n### 1. Calculating age of house\ntrain_df['Age_of_House'] = np.where(train_df['YearBuilt']==train_df['YearRemodAdd'],(train_df['YrSold']-train_df['YearBuilt']),((train_df['YrSold']-train_df['YearBuilt'])-(train_df['YrSold']-train_df['YearRemodAdd'])))\ntrain_df[['LotArea','Age_of_House','SalePrice']]\n","metadata":{"execution":{"iopub.status.busy":"2022-07-24T17:40:14.289466Z","iopub.execute_input":"2022-07-24T17:40:14.290018Z","iopub.status.idle":"2022-07-24T17:40:14.306023Z","shell.execute_reply.started":"2022-07-24T17:40:14.289984Z","shell.execute_reply":"2022-07-24T17:40:14.304852Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"##### plotting age & saleprice\nplt.figure(figsize=(12,8))\nplt.scatter(train_df['Age_of_House'],train_df['SalePrice'])\nplt.xlabel('Age of house')\nplt.ylabel('SalePrice')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-24T17:40:14.307564Z","iopub.execute_input":"2022-07-24T17:40:14.308072Z","iopub.status.idle":"2022-07-24T17:40:14.481003Z","shell.execute_reply.started":"2022-07-24T17:40:14.308039Z","shell.execute_reply":"2022-07-24T17:40:14.479300Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### as age of house increases, sale price of house is decreasing\n","metadata":{}},{"cell_type":"code","source":"#### dropping the year based columns,\ntrain_df.drop(['YearBuilt','YearRemodAdd','YrSold'],axis=1,inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-24T17:40:14.482978Z","iopub.execute_input":"2022-07-24T17:40:14.483423Z","iopub.status.idle":"2022-07-24T17:40:14.492383Z","shell.execute_reply.started":"2022-07-24T17:40:14.483386Z","shell.execute_reply":"2022-07-24T17:40:14.490788Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"numerical_cols_updated= [i for i in train_df.columns if i not in categorical_cols and i!='Id']\nnumerical_cols_updated","metadata":{"execution":{"iopub.status.busy":"2022-07-24T17:40:14.494111Z","iopub.execute_input":"2022-07-24T17:40:14.494532Z","iopub.status.idle":"2022-07-24T17:40:14.510259Z","shell.execute_reply.started":"2022-07-24T17:40:14.494473Z","shell.execute_reply":"2022-07-24T17:40:14.509300Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"### draw heatmap for correlation for updated train dataset\n\nplt.figure(figsize=(25,18))\nsb.heatmap(train_df[numerical_cols_updated].corr(),annot=True)\nplt.savefig('./data.png')","metadata":{"execution":{"iopub.status.busy":"2022-07-24T17:40:14.511541Z","iopub.execute_input":"2022-07-24T17:40:14.512331Z","iopub.status.idle":"2022-07-24T17:40:18.279421Z","shell.execute_reply.started":"2022-07-24T17:40:14.512295Z","shell.execute_reply":"2022-07-24T17:40:18.278050Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"##### Feature Selection/ Studying the importance of features","metadata":{}},{"cell_type":"code","source":"X= train_df.drop(['SalePrice'],axis=1)\ny= train_df['SalePrice']\n","metadata":{"execution":{"iopub.status.busy":"2022-07-24T17:40:18.280980Z","iopub.execute_input":"2022-07-24T17:40:18.281361Z","iopub.status.idle":"2022-07-24T17:40:18.289092Z","shell.execute_reply.started":"2022-07-24T17:40:18.281329Z","shell.execute_reply":"2022-07-24T17:40:18.287390Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-24T17:40:18.290453Z","iopub.execute_input":"2022-07-24T17:40:18.291034Z","iopub.status.idle":"2022-07-24T17:40:18.321593Z","shell.execute_reply.started":"2022-07-24T17:40:18.290987Z","shell.execute_reply":"2022-07-24T17:40:18.320558Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"numerical_cols_updated_features= [i for i in numerical_cols_updated if i!='SalePrice']\nnumerical_cols_updated_features","metadata":{"execution":{"iopub.status.busy":"2022-07-24T17:40:18.323217Z","iopub.execute_input":"2022-07-24T17:40:18.323607Z","iopub.status.idle":"2022-07-24T17:40:18.330907Z","shell.execute_reply.started":"2022-07-24T17:40:18.323574Z","shell.execute_reply":"2022-07-24T17:40:18.329584Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Studying the multicolinearity b/w features ","metadata":{}},{"cell_type":"code","source":"X_vif= X.copy()\n\nX_vif[numerical_cols_updated_features].skew()","metadata":{"execution":{"iopub.status.busy":"2022-07-24T17:40:18.332770Z","iopub.execute_input":"2022-07-24T17:40:18.334322Z","iopub.status.idle":"2022-07-24T17:40:18.349005Z","shell.execute_reply.started":"2022-07-24T17:40:18.334270Z","shell.execute_reply":"2022-07-24T17:40:18.347594Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_vif[numerical_cols_updated_features].describe()","metadata":{"execution":{"iopub.status.busy":"2022-07-24T17:40:18.350581Z","iopub.execute_input":"2022-07-24T17:40:18.351832Z","iopub.status.idle":"2022-07-24T17:40:18.427035Z","shell.execute_reply.started":"2022-07-24T17:40:18.351784Z","shell.execute_reply":"2022-07-24T17:40:18.426123Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":" #### scale the value via standerisation\nfrom sklearn.preprocessing import StandardScaler,MinMaxScaler\nsc_scaler= StandardScaler()\nX_vif[numerical_cols_updated_features]= sc_scaler.fit_transform(X_vif[numerical_cols_updated_features])\n","metadata":{"execution":{"iopub.status.busy":"2022-07-24T17:40:18.428374Z","iopub.execute_input":"2022-07-24T17:40:18.429062Z","iopub.status.idle":"2022-07-24T17:40:18.443117Z","shell.execute_reply.started":"2022-07-24T17:40:18.429024Z","shell.execute_reply":"2022-07-24T17:40:18.441899Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_vif[numerical_cols_updated_features].describe()","metadata":{"execution":{"iopub.status.busy":"2022-07-24T17:40:18.452527Z","iopub.execute_input":"2022-07-24T17:40:18.453164Z","iopub.status.idle":"2022-07-24T17:40:18.530638Z","shell.execute_reply.started":"2022-07-24T17:40:18.453119Z","shell.execute_reply":"2022-07-24T17:40:18.529310Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from statsmodels.stats.outliers_influence import variance_inflation_factor\nvif= pd.DataFrame()\nvif['Feature']= numerical_cols_updated_features\nvif['VIF']= [variance_inflation_factor(X_vif[numerical_cols_updated_features].values,i) for i in range(X_vif[numerical_cols_updated_features].shape[1])]\n\nvif","metadata":{"execution":{"iopub.status.busy":"2022-07-24T17:40:18.534216Z","iopub.execute_input":"2022-07-24T17:40:18.534625Z","iopub.status.idle":"2022-07-24T17:40:18.712687Z","shell.execute_reply.started":"2022-07-24T17:40:18.534589Z","shell.execute_reply":"2022-07-24T17:40:18.710768Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df[['BsmtFinSF1', 'BsmtFinSF2', 'BsmtUnfSF', 'TotalBsmtSF', '1stFlrSF', '2ndFlrSF', 'LowQualFinSF', 'GrLivArea','SalePrice']].corr()","metadata":{"execution":{"iopub.status.busy":"2022-07-24T17:40:18.719896Z","iopub.execute_input":"2022-07-24T17:40:18.723660Z","iopub.status.idle":"2022-07-24T17:40:18.760418Z","shell.execute_reply.started":"2022-07-24T17:40:18.723577Z","shell.execute_reply":"2022-07-24T17:40:18.759238Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"##### observations\n\n1. Features like BsmtFinSF1, BsmtFinSF2, BsmtUnfSF, TotalBsmtSF, 1stFlrSF, 2ndFlrSF, LowQualFinSF, GrLivArea are vif equal to infinity, which means they are highly multicolinear. \nAlso from correlation table above, it could be seen that some features like TotalBsmtSF, 1stFlrSF, GrLivArea are having good correlation with Saleprice, ie the target variable/feature\n\nlet remove the others and see what would be the value of vif of remaining","metadata":{}},{"cell_type":"code","source":"X_vif.drop(['BsmtFinSF1','BsmtFinSF2','BsmtUnfSF','2ndFlrSF', 'LowQualFinSF'],axis=1,inplace=True)\n\nfor i in ['BsmtFinSF1','BsmtFinSF2','BsmtUnfSF','2ndFlrSF', 'LowQualFinSF']:\n    numerical_cols_updated_features.remove(i)\n    \n\n","metadata":{"execution":{"iopub.status.busy":"2022-07-24T17:40:18.762215Z","iopub.execute_input":"2022-07-24T17:40:18.762948Z","iopub.status.idle":"2022-07-24T17:40:18.773997Z","shell.execute_reply.started":"2022-07-24T17:40:18.762901Z","shell.execute_reply":"2022-07-24T17:40:18.771673Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"vif= pd.DataFrame()\nvif['Feature']= numerical_cols_updated_features\nvif['VIF']= [variance_inflation_factor(X_vif[numerical_cols_updated_features].values,i) for i in range(X_vif[numerical_cols_updated_features].shape[1])]\n\nvif","metadata":{"execution":{"iopub.status.busy":"2022-07-24T17:40:18.776173Z","iopub.execute_input":"2022-07-24T17:40:18.777111Z","iopub.status.idle":"2022-07-24T17:40:18.890114Z","shell.execute_reply.started":"2022-07-24T17:40:18.777062Z","shell.execute_reply":"2022-07-24T17:40:18.888978Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"##### now vif of features seems to be fine :), eventhough there are some features having little high vif values but less than 5, we can ignore them now.","metadata":{}},{"cell_type":"code","source":"#### let's drop the COLUMNS/ Features from train_df\nX.drop(['Id','BsmtFinSF1','BsmtFinSF2','BsmtUnfSF','2ndFlrSF', 'LowQualFinSF'],axis=1,inplace=True)\nX\n\n","metadata":{"execution":{"iopub.status.busy":"2022-07-24T17:40:18.891870Z","iopub.execute_input":"2022-07-24T17:40:18.892574Z","iopub.status.idle":"2022-07-24T17:40:18.946525Z","shell.execute_reply.started":"2022-07-24T17:40:18.892529Z","shell.execute_reply":"2022-07-24T17:40:18.945388Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y","metadata":{"execution":{"iopub.status.busy":"2022-07-24T17:40:18.948183Z","iopub.execute_input":"2022-07-24T17:40:18.948883Z","iopub.status.idle":"2022-07-24T17:40:18.958450Z","shell.execute_reply.started":"2022-07-24T17:40:18.948840Z","shell.execute_reply":"2022-07-24T17:40:18.957288Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### let's end the feature selection part here, will do later based on model's performance","metadata":{}},{"cell_type":"code","source":"### splitting the dataset into train and test....\nfrom sklearn.model_selection import train_test_split\nX_train,X_test,y_train,y_test= train_test_split(X,y,test_size=0.1,random_state=0)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-24T17:40:18.960193Z","iopub.execute_input":"2022-07-24T17:40:18.960944Z","iopub.status.idle":"2022-07-24T17:40:18.975486Z","shell.execute_reply.started":"2022-07-24T17:40:18.960850Z","shell.execute_reply":"2022-07-24T17:40:18.974186Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"numerical_cols_updated_features","metadata":{"execution":{"iopub.status.busy":"2022-07-24T17:40:18.977142Z","iopub.execute_input":"2022-07-24T17:40:18.977852Z","iopub.status.idle":"2022-07-24T17:40:18.992491Z","shell.execute_reply.started":"2022-07-24T17:40:18.977808Z","shell.execute_reply":"2022-07-24T17:40:18.991198Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#### standerdizing the train and test dataset\n#sc_scaler= StandardScaler()\n#X_train[numerical_cols_updated_features]= sc_scaler.fit_transform(X_train[numerical_cols_updated_features])\n#X_test[numerical_cols_updated_features]= sc_scaler.transform(X_test[numerical_cols_updated_features])","metadata":{"execution":{"iopub.status.busy":"2022-07-24T17:40:18.994253Z","iopub.execute_input":"2022-07-24T17:40:18.994968Z","iopub.status.idle":"2022-07-24T17:40:19.007647Z","shell.execute_reply.started":"2022-07-24T17:40:18.994923Z","shell.execute_reply":"2022-07-24T17:40:19.006421Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in ['MSSubClass','OverallQual','OverallCond','BsmtFullBath','BsmtHalfBath','FullBath','HalfBath','BedroomAbvGr','TotRmsAbvGrd']:\n    X_train[i]= X_train[i].astype('int64')\n    X_test[i]= X_test[i].astype('int64')\n    ","metadata":{"execution":{"iopub.status.busy":"2022-07-24T17:40:19.009723Z","iopub.execute_input":"2022-07-24T17:40:19.010219Z","iopub.status.idle":"2022-07-24T17:40:19.038879Z","shell.execute_reply.started":"2022-07-24T17:40:19.010157Z","shell.execute_reply":"2022-07-24T17:40:19.037618Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"categorical_cols= [i for i in X.columns if X_train[i].dtype==object]\ncategorical_cols.remove('GarageYrBlt')","metadata":{"execution":{"iopub.status.busy":"2022-07-24T17:40:19.040563Z","iopub.execute_input":"2022-07-24T17:40:19.041323Z","iopub.status.idle":"2022-07-24T17:40:19.050099Z","shell.execute_reply.started":"2022-07-24T17:40:19.041275Z","shell.execute_reply":"2022-07-24T17:40:19.048778Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train[categorical_cols]","metadata":{"execution":{"iopub.status.busy":"2022-07-24T17:40:19.052187Z","iopub.execute_input":"2022-07-24T17:40:19.052684Z","iopub.status.idle":"2022-07-24T17:40:19.087392Z","shell.execute_reply.started":"2022-07-24T17:40:19.052637Z","shell.execute_reply":"2022-07-24T17:40:19.086247Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dict_1= {i:list(X[i].unique()) for i in categorical_cols}\ntemp1= pd.DataFrame.from_dict(dict_1, orient='index').T\ntemp1","metadata":{"execution":{"iopub.status.busy":"2022-07-24T17:40:19.089333Z","iopub.execute_input":"2022-07-24T17:40:19.089784Z","iopub.status.idle":"2022-07-24T17:40:19.133125Z","shell.execute_reply.started":"2022-07-24T17:40:19.089741Z","shell.execute_reply":"2022-07-24T17:40:19.131795Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#### label encoding the other object data\nfrom sklearn.preprocessing import LabelEncoder\nle_encoder_dict= {col:LabelEncoder() for col in temp1.columns}\nfor col in temp1.columns:\n    temp1[col]= le_encoder_dict[col].fit_transform(temp1[col])\ntemp1","metadata":{"execution":{"iopub.status.busy":"2022-07-24T17:40:19.135455Z","iopub.execute_input":"2022-07-24T17:40:19.136193Z","iopub.status.idle":"2022-07-24T17:40:19.177033Z","shell.execute_reply.started":"2022-07-24T17:40:19.136154Z","shell.execute_reply":"2022-07-24T17:40:19.176156Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for col in temp1.columns:\n    X_train[col]= le_encoder_dict[col].transform(X_train[col])\nX_train[categorical_cols]","metadata":{"execution":{"iopub.status.busy":"2022-07-24T17:40:19.177974Z","iopub.execute_input":"2022-07-24T17:40:19.178264Z","iopub.status.idle":"2022-07-24T17:40:19.232386Z","shell.execute_reply.started":"2022-07-24T17:40:19.178235Z","shell.execute_reply":"2022-07-24T17:40:19.231053Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train","metadata":{"execution":{"iopub.status.busy":"2022-07-24T17:40:19.233599Z","iopub.execute_input":"2022-07-24T17:40:19.233937Z","iopub.status.idle":"2022-07-24T17:40:19.256461Z","shell.execute_reply.started":"2022-07-24T17:40:19.233908Z","shell.execute_reply":"2022-07-24T17:40:19.255326Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for col in temp1.columns:\n    X_test[col]= le_encoder_dict[col].transform(X_test[col])\nX_test","metadata":{"execution":{"iopub.status.busy":"2022-07-24T17:40:19.258342Z","iopub.execute_input":"2022-07-24T17:40:19.258813Z","iopub.status.idle":"2022-07-24T17:40:19.297842Z","shell.execute_reply.started":"2022-07-24T17:40:19.258770Z","shell.execute_reply":"2022-07-24T17:40:19.296568Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def convert_year_numeric(x):\n    if x=='No Garage':\n        x=0\n    return x\nX_train['GarageYrBlt']= X_train['GarageYrBlt'].apply(convert_year_numeric)\nX_test['GarageYrBlt']= X_test['GarageYrBlt'].apply(convert_year_numeric)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-24T17:40:19.299860Z","iopub.execute_input":"2022-07-24T17:40:19.300302Z","iopub.status.idle":"2022-07-24T17:40:19.309315Z","shell.execute_reply.started":"2022-07-24T17:40:19.300258Z","shell.execute_reply":"2022-07-24T17:40:19.308157Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Model Applying -------------","metadata":{}},{"cell_type":"code","source":"import sklearn.metrics as sm\n#### define a function for score calculation regression\ndef model_score(y_test,y_pred,t):\n    mse= sm.mean_squared_log_error(y_test,y_pred)\n    mae= sm.mean_absolute_error(y_test,y_pred)\n    r2= sm.r2_score(y_test,y_pred)\n    rmse= np.sqrt(mse)\n    print(f'{t} log_mse: {mse}')\n    print(f'{t} mae: {mae}')\n    print(f'{t} r2: {r2}')\n    print(f'{t} log rmse: {rmse}')\n    return r2,rmse,mae\n    \n\n#### plotting the feature importance\n\ndef plot_feature_imp(feature_imp,features,model):\n    featureimp_df= pd.DataFrame({'Feature':features,'Feature Importance':feature_imp})\n    plt.figure(figsize=(18,16))\n    plt.title(f'feature importance graph for {model}')\n    plt.barh(y=featureimp_df['Feature'],width=featureimp_df['Feature Importance'])\n    plt.xlim(featureimp_df['Feature Importance'].min(),featureimp_df['Feature Importance'].max())\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-24T17:40:19.311018Z","iopub.execute_input":"2022-07-24T17:40:19.311591Z","iopub.status.idle":"2022-07-24T17:40:19.322171Z","shell.execute_reply.started":"2022-07-24T17:40:19.311555Z","shell.execute_reply":"2022-07-24T17:40:19.321330Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### 1. Decision Tree Regressor","metadata":{}},{"cell_type":"code","source":"from sklearn.tree import DecisionTreeRegressor\ndt_model= DecisionTreeRegressor(random_state=0)\ndt_model.fit(X_train,y_train)\nytest_pred= dt_model.predict(X_test)\nytrain_pred= dt_model.predict(X_train)\ndtreg_train_metrics=model_score(y_train,ytrain_pred,'train')\nprint('----------------------------------')\ndtreg_test_metrics=model_score(y_test,ytest_pred,'test')","metadata":{"execution":{"iopub.status.busy":"2022-07-24T17:40:19.323456Z","iopub.execute_input":"2022-07-24T17:40:19.323961Z","iopub.status.idle":"2022-07-24T17:40:19.534626Z","shell.execute_reply.started":"2022-07-24T17:40:19.323930Z","shell.execute_reply":"2022-07-24T17:40:19.533524Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_feature_imp(dt_model.feature_importances_,X_train.columns,'Decision Tree Model')","metadata":{"execution":{"iopub.status.busy":"2022-07-24T17:40:19.536092Z","iopub.execute_input":"2022-07-24T17:40:19.536567Z","iopub.status.idle":"2022-07-24T17:40:20.503305Z","shell.execute_reply.started":"2022-07-24T17:40:19.536522Z","shell.execute_reply":"2022-07-24T17:40:20.502348Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"##### observations\n\n1. Model is overfitting, train score is 1.0 while test score is 0.66.","metadata":{}},{"cell_type":"markdown","source":"##### Random Forest Regressor","metadata":{}},{"cell_type":"code","source":"from sklearn.ensemble import RandomForestRegressor\nrf_reg= RandomForestRegressor(random_state=1)\nrf_reg.fit(X_train,y_train)\nytest_pred= rf_reg.predict(X_test)\nytrain_pred= rf_reg.predict(X_train)\nrfreg_train_metrics=model_score(y_train,ytrain_pred,'train')\nprint('----------------------------------')\nrfreg_test_metrics=model_score(y_test,ytest_pred,'test')\n","metadata":{"execution":{"iopub.status.busy":"2022-07-24T17:40:20.504648Z","iopub.execute_input":"2022-07-24T17:40:20.505550Z","iopub.status.idle":"2022-07-24T17:40:22.246735Z","shell.execute_reply.started":"2022-07-24T17:40:20.505486Z","shell.execute_reply":"2022-07-24T17:40:22.245549Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### observations\n\n1. Little overfitting compared to decision tree ","metadata":{}},{"cell_type":"code","source":"plot_feature_imp(rf_reg.feature_importances_,X_train.columns,'Random Forest Regressor')","metadata":{"execution":{"iopub.status.busy":"2022-07-24T17:40:22.247972Z","iopub.execute_input":"2022-07-24T17:40:22.248284Z","iopub.status.idle":"2022-07-24T17:40:23.204058Z","shell.execute_reply.started":"2022-07-24T17:40:22.248256Z","shell.execute_reply":"2022-07-24T17:40:23.202238Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Adaboost Regressor","metadata":{}},{"cell_type":"code","source":"from sklearn.ensemble import AdaBoostRegressor\nadaboost_reg= AdaBoostRegressor(random_state=0)\nadaboost_reg.fit(X_train,y_train)\nytest_pred= adaboost_reg.predict(X_test)\nytrain_pred= adaboost_reg.predict(X_train)\nadaboostreg_train_metrics=model_score(y_train,ytrain_pred,'train')\nprint('----------------------------------')\nadaboostreg_test_metrics=model_score(y_test,ytest_pred,'test')\n","metadata":{"execution":{"iopub.status.busy":"2022-07-24T17:40:23.205928Z","iopub.execute_input":"2022-07-24T17:40:23.206541Z","iopub.status.idle":"2022-07-24T17:40:23.537580Z","shell.execute_reply.started":"2022-07-24T17:40:23.206479Z","shell.execute_reply":"2022-07-24T17:40:23.536422Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_feature_imp(adaboost_reg.feature_importances_,X_train.columns,'Adaboost Regressor')","metadata":{"execution":{"iopub.status.busy":"2022-07-24T17:40:23.540441Z","iopub.execute_input":"2022-07-24T17:40:23.542743Z","iopub.status.idle":"2022-07-24T17:40:24.489217Z","shell.execute_reply.started":"2022-07-24T17:40:23.542705Z","shell.execute_reply":"2022-07-24T17:40:24.488435Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"##### Observations\n\n1. Model seems to better than above, but has some overfitting too.need to see via tuning hyperparameters","metadata":{}},{"cell_type":"markdown","source":"#### GB Regressor","metadata":{}},{"cell_type":"code","source":"from sklearn.ensemble import GradientBoostingRegressor\ngb_reg= GradientBoostingRegressor(random_state=0)\ngb_reg.fit(X_train,y_train)\nytest_pred= gb_reg.predict(X_test)\nytrain_pred= gb_reg.predict(X_train)\ngbreg_train_metrics=model_score(y_train,ytrain_pred,'train')\nprint('----------------------------------')\ngbreg_test_metrics=model_score(y_test,ytest_pred,'test')\n","metadata":{"execution":{"iopub.status.busy":"2022-07-24T17:40:24.490599Z","iopub.execute_input":"2022-07-24T17:40:24.491122Z","iopub.status.idle":"2022-07-24T17:40:25.040112Z","shell.execute_reply.started":"2022-07-24T17:40:24.491087Z","shell.execute_reply":"2022-07-24T17:40:25.038944Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"##### observations\n\n1. Model seems to little overfitting, need to figure out via hyperparameter tuning","metadata":{}},{"cell_type":"code","source":"plot_feature_imp(gb_reg.feature_importances_,X_train.columns,'GB Regressor')","metadata":{"execution":{"iopub.status.busy":"2022-07-24T17:40:25.041408Z","iopub.execute_input":"2022-07-24T17:40:25.041775Z","iopub.status.idle":"2022-07-24T17:40:25.992704Z","shell.execute_reply.started":"2022-07-24T17:40:25.041744Z","shell.execute_reply":"2022-07-24T17:40:25.991907Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### XGB Regressor","metadata":{}},{"cell_type":"code","source":"from xgboost import XGBRegressor\nxgb_reg= XGBRegressor(random_state=1)\nxgb_reg.fit(X_train,y_train)\nytest_pred= xgb_reg.predict(X_test)\nytrain_pred= xgb_reg.predict(X_train)\nxgbreg_train_metrics=model_score(y_train,ytrain_pred,'train')\nprint('----------------------------------')\nxgbreg_test_metrics=model_score(y_test,ytest_pred,'test')\n","metadata":{"execution":{"iopub.status.busy":"2022-07-24T17:40:25.993864Z","iopub.execute_input":"2022-07-24T17:40:25.994730Z","iopub.status.idle":"2022-07-24T17:40:26.740525Z","shell.execute_reply.started":"2022-07-24T17:40:25.994695Z","shell.execute_reply":"2022-07-24T17:40:26.739373Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Observations\n\n1. Model seems to be overfitting a little","metadata":{}},{"cell_type":"code","source":"plot_feature_imp(xgb_reg.feature_importances_,X_train.columns,'XGB Regressor')","metadata":{"execution":{"iopub.status.busy":"2022-07-24T17:40:26.741661Z","iopub.execute_input":"2022-07-24T17:40:26.741966Z","iopub.status.idle":"2022-07-24T17:40:27.679004Z","shell.execute_reply.started":"2022-07-24T17:40:26.741939Z","shell.execute_reply":"2022-07-24T17:40:27.677926Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#### comparing metrics\ndef compare_metrics(models,metrics):\n    compare_df= pd.DataFrame({'Metrics':['R2 Score','Log RMSE','MAE']})\n    for i in range(len(models)):\n        compare_df[models[i]]= metrics[i]\n    return compare_df","metadata":{"execution":{"iopub.status.busy":"2022-07-24T17:40:27.680539Z","iopub.execute_input":"2022-07-24T17:40:27.680858Z","iopub.status.idle":"2022-07-24T17:40:27.687226Z","shell.execute_reply.started":"2022-07-24T17:40:27.680830Z","shell.execute_reply":"2022-07-24T17:40:27.685934Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"models=['Decision Tree','RandomForest','Adaboost','GB Regressor','XGB Regressor']\ntrain_metrics=[dtreg_train_metrics,rfreg_train_metrics,adaboostreg_train_metrics,gbreg_train_metrics,xgbreg_train_metrics]\ntest_metrics=[dtreg_test_metrics,rfreg_test_metrics,adaboostreg_test_metrics,gbreg_test_metrics,xgbreg_test_metrics]\ntrain_comparedf= compare_metrics(models,train_metrics)\ntest_comparedf= compare_metrics(models,test_metrics)","metadata":{"execution":{"iopub.status.busy":"2022-07-24T17:40:27.689023Z","iopub.execute_input":"2022-07-24T17:40:27.689622Z","iopub.status.idle":"2022-07-24T17:40:27.713697Z","shell.execute_reply.started":"2022-07-24T17:40:27.689577Z","shell.execute_reply":"2022-07-24T17:40:27.712431Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_comparedf","metadata":{"execution":{"iopub.status.busy":"2022-07-24T17:40:27.715481Z","iopub.execute_input":"2022-07-24T17:40:27.715908Z","iopub.status.idle":"2022-07-24T17:40:27.738548Z","shell.execute_reply.started":"2022-07-24T17:40:27.715875Z","shell.execute_reply":"2022-07-24T17:40:27.737616Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_comparedf","metadata":{"execution":{"iopub.status.busy":"2022-07-24T17:40:27.739986Z","iopub.execute_input":"2022-07-24T17:40:27.740555Z","iopub.status.idle":"2022-07-24T17:40:27.758729Z","shell.execute_reply.started":"2022-07-24T17:40:27.740492Z","shell.execute_reply":"2022-07-24T17:40:27.757808Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"##### Randomized Search CV -----","metadata":{}},{"cell_type":"code","source":"### FUNCTION to create the randomized hyperparameter tuning\nfrom sklearn.model_selection import RandomizedSearchCV\ndef random_hyperparameter_tuning(base_model,params,X_train,y_train):\n    random_model= RandomizedSearchCV(estimator=base_model,param_distributions=params,n_iter=10,scoring='neg_mean_squared_log_error',cv=5,random_state=0,error_score=0,return_train_score=True,verbose=10)\n    random_model.fit(X_train,y_train)\n    print()\n    print(f'The best model params: {random_model.best_params_}')\n    print(f'The best model score is: {random_model.best_score_}')\n    return random_model","metadata":{"execution":{"iopub.status.busy":"2022-07-24T17:40:27.759879Z","iopub.execute_input":"2022-07-24T17:40:27.760338Z","iopub.status.idle":"2022-07-24T17:40:27.773186Z","shell.execute_reply.started":"2022-07-24T17:40:27.760301Z","shell.execute_reply":"2022-07-24T17:40:27.772369Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"### tuning for decision tree\nparams={'max_depth':[i for i in range(3,15,3)],'min_samples_split':[i for i in range(10,100,20)],'min_samples_leaf':[i for i in range(10,50,10)]}\ndt_reg_randommodel= random_hyperparameter_tuning(dt_model,params,X_train,y_train)","metadata":{"execution":{"iopub.status.busy":"2022-07-24T17:40:27.774312Z","iopub.execute_input":"2022-07-24T17:40:27.774856Z","iopub.status.idle":"2022-07-24T17:40:28.622187Z","shell.execute_reply.started":"2022-07-24T17:40:27.774823Z","shell.execute_reply":"2022-07-24T17:40:28.621029Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#### tuning the random forest regressor\nparams={'n_estimators':[i for i in range(50,600,100)],'max_depth':[i for i in range(3,15,3)],'min_samples_split':[i for i in range(10,100,20)],'min_samples_leaf':[i for i in range(10,50,10)]}\nrf_reg_randommodel= random_hyperparameter_tuning(rf_reg,params,X_train,y_train)","metadata":{"execution":{"iopub.status.busy":"2022-07-24T17:40:28.623905Z","iopub.execute_input":"2022-07-24T17:40:28.624598Z","iopub.status.idle":"2022-07-24T17:42:14.036263Z","shell.execute_reply.started":"2022-07-24T17:40:28.624553Z","shell.execute_reply":"2022-07-24T17:42:14.034984Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#### tuning adaboost_reg\nparams= {'n_estimators':[i for i in range(50,550,50)],'base_estimator':[DecisionTreeRegressor(max_depth=i) for i in range(3,15,3)],'learning_rate':[0.2,0.4,0.6,0.8,0.9]}\nadaboost_reg_randommodel= random_hyperparameter_tuning(adaboost_reg,params,X_train,y_train)","metadata":{"execution":{"iopub.status.busy":"2022-07-24T17:42:14.037570Z","iopub.execute_input":"2022-07-24T17:42:14.037867Z","iopub.status.idle":"2022-07-24T17:44:47.219940Z","shell.execute_reply.started":"2022-07-24T17:42:14.037840Z","shell.execute_reply":"2022-07-24T17:44:47.218520Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#### tuning the GB model,\nparams ={'n_estimators':[i for i in range(50,550,50)],'learning_rate':[0.1,0.3,0.5,0.7,0.9,1.0],'min_samples_split':[i for i in range(10,100,20)],'min_samples_leaf':[i for i in range(10,100,15)],'max_depth':[i for i in range(3,18,3)]}\ngb_reg_randommodel=random_hyperparameter_tuning(gb_reg,params,X_train,y_train)","metadata":{"execution":{"iopub.status.busy":"2022-07-24T17:44:47.221704Z","iopub.execute_input":"2022-07-24T17:44:47.222702Z","iopub.status.idle":"2022-07-24T17:46:54.042455Z","shell.execute_reply.started":"2022-07-24T17:44:47.222656Z","shell.execute_reply":"2022-07-24T17:46:54.041313Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#### xgb regressor\nparams={'eta':[0.3,0.5,0.6,0.9],'gamma':[i for i in range(5,20)],'max_depth':[i for i in range(3,20,2)]}\nxgb_reg_randommodel= random_hyperparameter_tuning(xgb_reg,params,X_train,y_train)","metadata":{"execution":{"iopub.status.busy":"2022-07-24T17:46:54.043841Z","iopub.execute_input":"2022-07-24T17:46:54.044169Z","iopub.status.idle":"2022-07-24T17:47:52.320275Z","shell.execute_reply.started":"2022-07-24T17:46:54.044140Z","shell.execute_reply":"2022-07-24T17:47:52.319524Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#### TRYING with updated models\n#Decision Tree\ndt_model= DecisionTreeRegressor(**dt_reg_randommodel.best_params_,random_state=0)\ndt_model.fit(X_train,y_train)\nytest_pred= dt_model.predict(X_test)\nytrain_pred= dt_model.predict(X_train)\ndtreg_train_metrics=model_score(y_train,ytrain_pred,'train')\nprint('----------------------------------')\ndtreg_test_metrics=model_score(y_test,ytest_pred,'test')","metadata":{"execution":{"iopub.status.busy":"2022-07-24T17:47:52.321486Z","iopub.execute_input":"2022-07-24T17:47:52.321833Z","iopub.status.idle":"2022-07-24T17:47:52.348118Z","shell.execute_reply.started":"2022-07-24T17:47:52.321803Z","shell.execute_reply":"2022-07-24T17:47:52.347097Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Random forest regressor\nrf_reg= RandomForestRegressor(**rf_reg_randommodel.best_params_,random_state=1)\nrf_reg.fit(X_train,y_train)\nytest_pred= rf_reg.predict(X_test)\nytrain_pred= rf_reg.predict(X_train)\nrfreg_train_metrics=model_score(y_train,ytrain_pred,'train')\nprint('----------------------------------')\nrfreg_test_metrics=model_score(y_test,ytest_pred,'test')","metadata":{"execution":{"iopub.status.busy":"2022-07-24T17:47:52.349851Z","iopub.execute_input":"2022-07-24T17:47:52.350279Z","iopub.status.idle":"2022-07-24T17:47:55.781375Z","shell.execute_reply.started":"2022-07-24T17:47:52.350237Z","shell.execute_reply":"2022-07-24T17:47:55.780283Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"### adaboost regressor\n\nadaboost_reg= AdaBoostRegressor(**adaboost_reg_randommodel.best_params_,random_state=0)\nadaboost_reg.fit(X_train,y_train)\nytest_pred= adaboost_reg.predict(X_test)\nytrain_pred= adaboost_reg.predict(X_train)\nadaboostreg_train_metrics=model_score(y_train,ytrain_pred,'train')\nprint('----------------------------------')\nadaboostreg_test_metrics=model_score(y_test,ytest_pred,'test')","metadata":{"execution":{"iopub.status.busy":"2022-07-24T17:47:55.782905Z","iopub.execute_input":"2022-07-24T17:47:55.783212Z","iopub.status.idle":"2022-07-24T17:47:58.578251Z","shell.execute_reply.started":"2022-07-24T17:47:55.783185Z","shell.execute_reply":"2022-07-24T17:47:58.577067Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# gb regressor\ngb_reg= GradientBoostingRegressor(**gb_reg_randommodel.best_params_,random_state=0)\ngb_reg.fit(X_train,y_train)\nytest_pred= gb_reg.predict(X_test)\nytrain_pred= gb_reg.predict(X_train)\ngbreg_train_metrics=model_score(y_train,ytrain_pred,'train')\nprint('----------------------------------')\ngbreg_test_metrics=model_score(y_test,ytest_pred,'test')","metadata":{"execution":{"iopub.status.busy":"2022-07-24T17:47:58.580002Z","iopub.execute_input":"2022-07-24T17:47:58.580423Z","iopub.status.idle":"2022-07-24T17:48:02.254749Z","shell.execute_reply.started":"2022-07-24T17:47:58.580381Z","shell.execute_reply":"2022-07-24T17:48:02.253410Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"### XGB Regressor\nxgb_reg= XGBRegressor(**xgb_reg_randommodel.best_params_,random_state=1)\nxgb_reg.fit(X_train,y_train)\nytest_pred= xgb_reg.predict(X_test)\nytrain_pred= xgb_reg.predict(X_train)\nxgbreg_train_metrics=model_score(y_train,ytrain_pred,'train')\nprint('----------------------------------')\nxgbreg_test_metrics=model_score(y_test,ytest_pred,'test')","metadata":{"execution":{"iopub.status.busy":"2022-07-24T17:48:02.257926Z","iopub.execute_input":"2022-07-24T17:48:02.258288Z","iopub.status.idle":"2022-07-24T17:48:02.682570Z","shell.execute_reply.started":"2022-07-24T17:48:02.258257Z","shell.execute_reply":"2022-07-24T17:48:02.681146Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"models=['Decision Tree','RandomForest','Adaboost','GB Regressor','XGB Regressor']\ntrain_metrics=[dtreg_train_metrics,rfreg_train_metrics,adaboostreg_train_metrics,gbreg_train_metrics,xgbreg_train_metrics]\ntest_metrics=[dtreg_test_metrics,rfreg_test_metrics,adaboostreg_test_metrics,gbreg_test_metrics,xgbreg_test_metrics]\ntrain_comparedf= compare_metrics(models,train_metrics)\ntest_comparedf= compare_metrics(models,test_metrics)","metadata":{"execution":{"iopub.status.busy":"2022-07-24T17:48:02.683805Z","iopub.execute_input":"2022-07-24T17:48:02.684277Z","iopub.status.idle":"2022-07-24T17:48:02.695439Z","shell.execute_reply.started":"2022-07-24T17:48:02.684247Z","shell.execute_reply":"2022-07-24T17:48:02.694197Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_comparedf","metadata":{"execution":{"iopub.status.busy":"2022-07-24T17:48:02.696679Z","iopub.execute_input":"2022-07-24T17:48:02.697445Z","iopub.status.idle":"2022-07-24T17:48:02.711559Z","shell.execute_reply.started":"2022-07-24T17:48:02.697409Z","shell.execute_reply":"2022-07-24T17:48:02.710745Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_comparedf","metadata":{"execution":{"iopub.status.busy":"2022-07-24T17:48:02.712719Z","iopub.execute_input":"2022-07-24T17:48:02.713348Z","iopub.status.idle":"2022-07-24T17:48:02.726004Z","shell.execute_reply.started":"2022-07-24T17:48:02.713315Z","shell.execute_reply":"2022-07-24T17:48:02.724811Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### from this observations, Random forest regressor is better\n\n1. R2 is balanced and is a good value (0.88)\n\n2. RMSE , MAE are both balanced.\n\n3. Next comes Decision Tree , adaboost and gb regressor ","metadata":{}},{"cell_type":"code","source":"test_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-24T17:49:09.890168Z","iopub.execute_input":"2022-07-24T17:49:09.890612Z","iopub.status.idle":"2022-07-24T17:49:09.918619Z","shell.execute_reply.started":"2022-07-24T17:49:09.890572Z","shell.execute_reply":"2022-07-24T17:49:09.917771Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df.isnull().sum()[test_df.isnull().sum()!=0]","metadata":{"execution":{"iopub.status.busy":"2022-07-24T17:49:42.101037Z","iopub.execute_input":"2022-07-24T17:49:42.101558Z","iopub.status.idle":"2022-07-24T17:49:42.127805Z","shell.execute_reply.started":"2022-07-24T17:49:42.101495Z","shell.execute_reply":"2022-07-24T17:49:42.126594Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df.drop('LotFrontage',axis=1,inplace=True)\ntest_df['Alley'].fillna('No Alley Access',inplace=True)\nfor c in ['BsmtQual','BsmtCond','BsmtExposure','BsmtFinType1','BsmtFinType2']:\n    test_df[c].fillna('No Basement',inplace=True)\n    \n\ntest_df['FireplaceQu'].fillna('No fireplace',inplace=True)\nfor c1 in ['GarageType','GarageYrBlt','GarageFinish','GarageQual','GarageCond']:\n    test_df[c1].fillna('No Garage',inplace=True)\n    \n\ntest_df['PoolQC'].fillna('No Pool',inplace=True)\ntest_df['Fence'].fillna('No Fence',inplace=True)\ntest_df['MiscFeature'].fillna('No Misc Feature',inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-24T17:52:41.502103Z","iopub.execute_input":"2022-07-24T17:52:41.502471Z","iopub.status.idle":"2022-07-24T17:52:41.521459Z","shell.execute_reply.started":"2022-07-24T17:52:41.502442Z","shell.execute_reply":"2022-07-24T17:52:41.520059Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df.isnull().sum()[test_df.isnull().sum()!=0]","metadata":{"execution":{"iopub.status.busy":"2022-07-24T17:52:58.267809Z","iopub.execute_input":"2022-07-24T17:52:58.268179Z","iopub.status.idle":"2022-07-24T17:52:58.287285Z","shell.execute_reply.started":"2022-07-24T17:52:58.268150Z","shell.execute_reply":"2022-07-24T17:52:58.286099Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df['MSZoning'].fillna(test_df['MSZoning'].mode()[0],inplace=True)\ntest_df['Utilities'].fillna(test_df['Utilities'].mode()[0],inplace=True)\ntest_df['Exterior1st'].fillna(test_df['Exterior1st'].mode()[0],inplace=True)\ntest_df['Exterior2nd'].fillna(test_df['Exterior2nd'].mode()[0],inplace=True)\ntest_df['MasVnrType'].fillna(test_df['MasVnrType'].mode()[0],inplace=True)\ntest_df['MasVnrArea'].fillna(test_df['MasVnrArea'].mean(),inplace=True)\ntest_df['BsmtFinSF1'].fillna(test_df['BsmtFinSF1'].mean(),inplace=True)\ntest_df['BsmtFinSF2'].fillna(test_df['BsmtFinSF2'].mean(),inplace=True)\ntest_df['BsmtUnfSF'].fillna(test_df['BsmtUnfSF'].mean(),inplace=True)\ntest_df['TotalBsmtSF'].fillna(test_df['TotalBsmtSF'].mean(),inplace=True)\ntest_df['BsmtFullBath'].fillna(test_df['BsmtFullBath'].mode()[0],inplace=True)\ntest_df['BsmtHalfBath'].fillna(test_df['BsmtHalfBath'].mode()[0],inplace=True)\ntest_df['KitchenQual'].fillna(test_df['KitchenQual'].mode()[0],inplace=True)\ntest_df['Functional'].fillna(test_df['Functional'].mode()[0],inplace=True)\ntest_df['GarageCars'].fillna(test_df['GarageCars'].mode()[0],inplace=True)\ntest_df['GarageArea'].fillna(test_df['GarageArea'].mean(),inplace=True)\ntest_df['SaleType'].fillna(test_df['SaleType'].mode()[0],inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-24T18:18:34.452253Z","iopub.execute_input":"2022-07-24T18:18:34.452624Z","iopub.status.idle":"2022-07-24T18:18:34.480644Z","shell.execute_reply.started":"2022-07-24T18:18:34.452594Z","shell.execute_reply":"2022-07-24T18:18:34.479123Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for c in ['MSSubClass','OverallQual','OverallCond','BsmtFullBath','BsmtHalfBath','FullBath','HalfBath','BedroomAbvGr','TotRmsAbvGrd']:\n    test_df[c]= test_df[c].astype(object)\n\ntest_df['Age_of_House'] = np.where(test_df['YearBuilt']==test_df['YearRemodAdd'],(test_df['YrSold']-test_df['YearBuilt']),((test_df['YrSold']-test_df['YearBuilt'])-(test_df['YrSold']-test_df['YearRemodAdd'])))\ntest_df.drop(['YearBuilt','YearRemodAdd','YrSold'],axis=1,inplace=True)\n\ntest_df.drop(['BsmtFinSF1','BsmtFinSF2','BsmtUnfSF','2ndFlrSF', 'LowQualFinSF'],axis=1,inplace=True)\nfor i in ['MSSubClass','OverallQual','OverallCond','BsmtFullBath','BsmtHalfBath','FullBath','HalfBath','BedroomAbvGr','TotRmsAbvGrd']:\n    test_df[i]= test_df[i].astype('int64')","metadata":{"execution":{"iopub.status.busy":"2022-07-24T18:21:13.004043Z","iopub.execute_input":"2022-07-24T18:21:13.004430Z","iopub.status.idle":"2022-07-24T18:21:13.035364Z","shell.execute_reply.started":"2022-07-24T18:21:13.004401Z","shell.execute_reply":"2022-07-24T18:21:13.034019Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for col in temp1.columns:\n    test_df[col]= le_encoder_dict[col].transform(test_df[col])\ntest_df","metadata":{"execution":{"iopub.status.busy":"2022-07-24T18:21:51.944258Z","iopub.execute_input":"2022-07-24T18:21:51.945420Z","iopub.status.idle":"2022-07-24T18:21:52.009483Z","shell.execute_reply.started":"2022-07-24T18:21:51.945365Z","shell.execute_reply":"2022-07-24T18:21:52.008352Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df['GarageYrBlt']= test_df['GarageYrBlt'].apply(convert_year_numeric)","metadata":{"execution":{"iopub.status.busy":"2022-07-24T18:22:29.619908Z","iopub.execute_input":"2022-07-24T18:22:29.620312Z","iopub.status.idle":"2022-07-24T18:22:29.628786Z","shell.execute_reply.started":"2022-07-24T18:22:29.620281Z","shell.execute_reply":"2022-07-24T18:22:29.627341Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-24T18:22:48.105454Z","iopub.execute_input":"2022-07-24T18:22:48.105937Z","iopub.status.idle":"2022-07-24T18:22:48.126591Z","shell.execute_reply.started":"2022-07-24T18:22:48.105895Z","shell.execute_reply":"2022-07-24T18:22:48.125609Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-24T18:23:03.523002Z","iopub.execute_input":"2022-07-24T18:23:03.523365Z","iopub.status.idle":"2022-07-24T18:23:03.531647Z","shell.execute_reply.started":"2022-07-24T18:23:03.523337Z","shell.execute_reply":"2022-07-24T18:23:03.530422Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ids= test_df['Id']","metadata":{"execution":{"iopub.status.busy":"2022-07-24T18:23:59.275563Z","iopub.execute_input":"2022-07-24T18:23:59.275969Z","iopub.status.idle":"2022-07-24T18:23:59.281080Z","shell.execute_reply.started":"2022-07-24T18:23:59.275924Z","shell.execute_reply":"2022-07-24T18:23:59.279902Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df.drop('Id',axis=1,inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-24T18:24:20.583490Z","iopub.execute_input":"2022-07-24T18:24:20.583967Z","iopub.status.idle":"2022-07-24T18:24:20.590904Z","shell.execute_reply.started":"2022-07-24T18:24:20.583930Z","shell.execute_reply":"2022-07-24T18:24:20.589702Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred_final=rf_reg.predict(test_df)","metadata":{"execution":{"iopub.status.busy":"2022-07-24T18:46:21.194590Z","iopub.execute_input":"2022-07-24T18:46:21.194971Z","iopub.status.idle":"2022-07-24T18:46:21.219177Z","shell.execute_reply.started":"2022-07-24T18:46:21.194941Z","shell.execute_reply":"2022-07-24T18:46:21.218245Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"final_df= pd.DataFrame({'Id':ids,'SalePrice':y_pred_final})\nfinal_df","metadata":{"execution":{"iopub.status.busy":"2022-07-24T18:46:22.661319Z","iopub.execute_input":"2022-07-24T18:46:22.661718Z","iopub.status.idle":"2022-07-24T18:46:22.675939Z","shell.execute_reply.started":"2022-07-24T18:46:22.661676Z","shell.execute_reply":"2022-07-24T18:46:22.674815Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"final_df.to_csv('sample_submission.csv',index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-24T18:46:23.926019Z","iopub.execute_input":"2022-07-24T18:46:23.926741Z","iopub.status.idle":"2022-07-24T18:46:23.937394Z","shell.execute_reply.started":"2022-07-24T18:46:23.926695Z","shell.execute_reply":"2022-07-24T18:46:23.936361Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}