{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-19T12:37:10.274912Z","iopub.execute_input":"2022-07-19T12:37:10.275326Z","iopub.status.idle":"2022-07-19T12:37:10.285736Z","shell.execute_reply.started":"2022-07-19T12:37:10.275284Z","shell.execute_reply":"2022-07-19T12:37:10.284371Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Importing the libraries","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nsns.set()","metadata":{"execution":{"iopub.status.busy":"2022-07-19T12:37:13.089611Z","iopub.execute_input":"2022-07-19T12:37:13.089970Z","iopub.status.idle":"2022-07-19T12:37:13.730822Z","shell.execute_reply.started":"2022-07-19T12:37:13.089942Z","shell.execute_reply":"2022-07-19T12:37:13.729610Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# loading the data from csv file\n\ndf_train = pd.read_csv('../input/house-prices-advanced-regression-techniques/train.csv')\ndf_test = pd.read_csv('../input/house-prices-advanced-regression-techniques/test.csv')","metadata":{"execution":{"iopub.status.busy":"2022-07-19T12:37:13.735189Z","iopub.execute_input":"2022-07-19T12:37:13.735609Z","iopub.status.idle":"2022-07-19T12:37:13.812535Z","shell.execute_reply.started":"2022-07-19T12:37:13.735573Z","shell.execute_reply":"2022-07-19T12:37:13.811271Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# first 5 row of the train Data\n\ndf_train.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-19T12:37:13.938773Z","iopub.execute_input":"2022-07-19T12:37:13.939165Z","iopub.status.idle":"2022-07-19T12:37:13.979957Z","shell.execute_reply.started":"2022-07-19T12:37:13.939129Z","shell.execute_reply":"2022-07-19T12:37:13.978987Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# no of rows and columns in train data\n\ndf_train.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-19T12:37:23.145620Z","iopub.execute_input":"2022-07-19T12:37:23.146016Z","iopub.status.idle":"2022-07-19T12:37:23.153449Z","shell.execute_reply.started":"2022-07-19T12:37:23.145981Z","shell.execute_reply":"2022-07-19T12:37:23.152281Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# first 5 row of the test Data\n\ndf_test.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-19T12:37:30.296718Z","iopub.execute_input":"2022-07-19T12:37:30.297916Z","iopub.status.idle":"2022-07-19T12:37:30.324315Z","shell.execute_reply.started":"2022-07-19T12:37:30.297864Z","shell.execute_reply":"2022-07-19T12:37:30.323165Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# no of rows and columns in test data\n\ndf_test.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-19T12:37:37.614524Z","iopub.execute_input":"2022-07-19T12:37:37.615935Z","iopub.status.idle":"2022-07-19T12:37:37.623387Z","shell.execute_reply.started":"2022-07-19T12:37:37.615877Z","shell.execute_reply":"2022-07-19T12:37:37.622232Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# showing all the columns in train data\n\ndf_train.columns","metadata":{"execution":{"iopub.status.busy":"2022-07-19T12:37:43.403230Z","iopub.execute_input":"2022-07-19T12:37:43.403625Z","iopub.status.idle":"2022-07-19T12:37:43.410981Z","shell.execute_reply.started":"2022-07-19T12:37:43.403595Z","shell.execute_reply":"2022-07-19T12:37:43.410206Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# info of the data\n\ndf_train.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-19T12:37:49.943296Z","iopub.execute_input":"2022-07-19T12:37:49.943674Z","iopub.status.idle":"2022-07-19T12:37:49.982629Z","shell.execute_reply.started":"2022-07-19T12:37:49.943644Z","shell.execute_reply":"2022-07-19T12:37:49.981646Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# checking the description of data\n\ndf_train.describe()","metadata":{"execution":{"iopub.status.busy":"2022-07-19T12:38:00.273281Z","iopub.execute_input":"2022-07-19T12:38:00.273685Z","iopub.status.idle":"2022-07-19T12:38:00.385435Z","shell.execute_reply.started":"2022-07-19T12:38:00.273652Z","shell.execute_reply":"2022-07-19T12:38:00.384242Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# checking those columns which has missing values\n\ncol = []\nfor column in df_train.columns:\n    if df_train[column].isnull().sum() > 0:\n        col.append(column)\n\ncol","metadata":{"execution":{"iopub.status.busy":"2022-07-19T12:38:08.042733Z","iopub.execute_input":"2022-07-19T12:38:08.043108Z","iopub.status.idle":"2022-07-19T12:38:08.077494Z","shell.execute_reply.started":"2022-07-19T12:38:08.043078Z","shell.execute_reply":"2022-07-19T12:38:08.076361Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# checking the no & %age of missing values in column\n\ndef missingCounts():\n    for i in col:\n        print(f\"no of missing vlaues in {i} out of 1460 :- \", df_train[i].isnull().sum())\n        print(f'percentage of missing values in {i}',df_train[i].isnull().sum() / len(df_train[i]))\n        print('*------------------------------------------------------------*')\n        \nmissingCounts()\n        ","metadata":{"execution":{"iopub.status.busy":"2022-07-19T12:38:14.804399Z","iopub.execute_input":"2022-07-19T12:38:14.804794Z","iopub.status.idle":"2022-07-19T12:38:14.826963Z","shell.execute_reply.started":"2022-07-19T12:38:14.804761Z","shell.execute_reply":"2022-07-19T12:38:14.825516Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Observations:\n1. PoolQC, Fence, MiscFeatures, Alley has very less data so we can drop these column\n2. FireplaceQu also has less than 50% data missing, ordinal category, we can drop these column because mode will not work here.\n\n","metadata":{}},{"cell_type":"code","source":"# drop the columns for test and train \n\ndf_train.drop(columns=['PoolQC', 'Fence', 'MiscFeature', 'Alley','FireplaceQu'], axis=1, inplace=True)\ndf_test.drop(columns=['PoolQC', 'Fence', 'MiscFeature', 'Alley','FireplaceQu'], axis=1, inplace=True)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-19T12:38:45.516826Z","iopub.execute_input":"2022-07-19T12:38:45.517790Z","iopub.status.idle":"2022-07-19T12:38:45.527472Z","shell.execute_reply.started":"2022-07-19T12:38:45.517744Z","shell.execute_reply":"2022-07-19T12:38:45.526283Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# checking the no of column after deleting \n\nprint(df_train.shape)\nprint(df_test.shape)","metadata":{"execution":{"iopub.status.busy":"2022-07-19T12:38:50.783968Z","iopub.execute_input":"2022-07-19T12:38:50.785010Z","iopub.status.idle":"2022-07-19T12:38:50.791392Z","shell.execute_reply.started":"2022-07-19T12:38:50.784966Z","shell.execute_reply":"2022-07-19T12:38:50.789677Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# checking all the numerical columns\n\nprint(df_train.dtypes[df_train.dtypes == 'int64'])\nlen(df_train.dtypes[df_train.dtypes == 'int64'])","metadata":{"execution":{"iopub.status.busy":"2022-07-19T12:38:58.507013Z","iopub.execute_input":"2022-07-19T12:38:58.507940Z","iopub.status.idle":"2022-07-19T12:38:58.518898Z","shell.execute_reply.started":"2022-07-19T12:38:58.507898Z","shell.execute_reply":"2022-07-19T12:38:58.518093Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# let's see how data is distributed for every numerical column\n#fig, ax = plt.subplots()\n#df_train.plot(kind='scatter', x='MSSubClass', y = 'SalePrice')\n#plt.show()\n\n\nnumeric_data = []\nfor i in df_train.columns:\n    if df_train[i].dtypes == 'int64':\n        numeric_data.append(i)\n\nplt.figure(figsize=(20,25), facecolor='white')\nplotnumber=1\n\nfor column in numeric_data:\n    if plotnumber<=35:\n        ax = plt.subplot(8,5,plotnumber)\n        sns.distplot(df_train[column])\n        plt.xlabel(column, fontsize=20)\n    plotnumber+=1\nplt.tight_layout()\nplt.show()\n    ","metadata":{"execution":{"iopub.status.busy":"2022-07-19T12:39:05.938878Z","iopub.execute_input":"2022-07-19T12:39:05.939273Z","iopub.status.idle":"2022-07-19T12:39:15.195291Z","shell.execute_reply.started":"2022-07-19T12:39:05.939237Z","shell.execute_reply":"2022-07-19T12:39:15.194002Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# visualizing the relationship between the numerical features and the response using scatterplot\n\nplt.figure(figsize=(20,30), facecolor='white')\nplotnumber = 1\n\nfor column in numeric_data:\n    if plotnumber<=34 :\n        ax = plt.subplot(9,4,plotnumber)\n        plt.scatter(df_train[column],df_train['SalePrice'])\n        plt.xlabel(column,fontsize=20)\n        plt.ylabel('SalePrice',fontsize=20)\n    plotnumber+=1\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-19T12:39:26.876696Z","iopub.execute_input":"2022-07-19T12:39:26.877121Z","iopub.status.idle":"2022-07-19T12:39:34.113441Z","shell.execute_reply.started":"2022-07-19T12:39:26.877086Z","shell.execute_reply":"2022-07-19T12:39:34.112037Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# filling the missing values in numerical columns in test and train data\n\n# find those numerical columns which has missing values in train data\nfor i in df_train.columns:\n    if df_train[i].dtypes == 'int64' or df_train[i].dtypes == 'float64':\n        if df_train[i].isnull().sum() > 0:\n            print(i, \"-\", df_train[i].isnull().sum())","metadata":{"execution":{"iopub.status.busy":"2022-07-19T12:39:40.128157Z","iopub.execute_input":"2022-07-19T12:39:40.128556Z","iopub.status.idle":"2022-07-19T12:39:40.146269Z","shell.execute_reply.started":"2022-07-19T12:39:40.128524Z","shell.execute_reply":"2022-07-19T12:39:40.145379Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train['LotFrontage'] = df_train['LotFrontage'].fillna(df_train['LotFrontage'].mode()[0]).astype('int64')\ndf_train['MasVnrArea'] = df_train['MasVnrArea'].fillna(df_train['MasVnrArea'].mode()[0]).astype('int64')\ndf_train['GarageYrBlt'] = df_train['GarageYrBlt'].fillna(df_train['GarageYrBlt'].mode()[0]).astype('int64')","metadata":{"execution":{"iopub.status.busy":"2022-07-19T12:39:51.073142Z","iopub.execute_input":"2022-07-19T12:39:51.073555Z","iopub.status.idle":"2022-07-19T12:39:51.088099Z","shell.execute_reply.started":"2022-07-19T12:39:51.073522Z","shell.execute_reply":"2022-07-19T12:39:51.087179Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test['LotFrontage'] = df_test['LotFrontage'].fillna(df_test['LotFrontage'].mode()[0]).astype('int64')\ndf_test['MasVnrArea'] = df_test['MasVnrArea'].fillna(df_test['MasVnrArea'].mode()[0]).astype('int64')\ndf_test['GarageYrBlt'] = df_test['GarageYrBlt'].fillna(df_test['GarageYrBlt'].mode()[0]).astype('int64')","metadata":{"execution":{"iopub.status.busy":"2022-07-19T12:39:58.246736Z","iopub.execute_input":"2022-07-19T12:39:58.247144Z","iopub.status.idle":"2022-07-19T12:39:58.257821Z","shell.execute_reply.started":"2022-07-19T12:39:58.247110Z","shell.execute_reply":"2022-07-19T12:39:58.256903Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# finding the numerical column  which has missing values in test data\nfor i in df_test.columns:\n    if df_test[i].dtypes == 'int64' or df_test[i].dtypes == 'float64':\n        if df_test[i].isnull().sum() > 0:\n            print(i, \"-\", df_test[i].isnull().sum())","metadata":{"execution":{"iopub.status.busy":"2022-07-19T12:40:04.823814Z","iopub.execute_input":"2022-07-19T12:40:04.824239Z","iopub.status.idle":"2022-07-19T12:40:04.846230Z","shell.execute_reply.started":"2022-07-19T12:40:04.824177Z","shell.execute_reply":"2022-07-19T12:40:04.845282Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# changing all the float columns in test data as we don't have any float col in train data into int\n\ndf_test['BsmtFinSF1'] = df_test['BsmtFinSF1'].fillna(df_test['BsmtFinSF1'].mode()[0]).astype('int64')\ndf_test['BsmtFinSF2'] = df_test['BsmtFinSF2'].fillna(df_test['BsmtFinSF2'].mode()[0]).astype('int64')\ndf_test['BsmtUnfSF'] = df_test['BsmtUnfSF'].fillna(df_test['BsmtUnfSF'].mean()).astype('int64')\ndf_test['TotalBsmtSF'] = df_test['TotalBsmtSF'].fillna(df_test['TotalBsmtSF'].mean()).astype('int64')\ndf_test['BsmtFullBath'] = df_test['BsmtFullBath'].fillna(df_test['BsmtFullBath'].mean()).astype('int64')\ndf_test['BsmtHalfBath'] = df_test['BsmtHalfBath'].fillna(df_test['BsmtHalfBath'].mode()[0]).astype('int64')\ndf_test['GarageCars'] = df_test['GarageCars'].fillna(df_test['GarageCars'].mode()[0]).astype('int64')\ndf_test['GarageArea'] = df_test['GarageArea'].fillna(df_test['GarageArea'].mode()[0]).astype('int64')","metadata":{"execution":{"iopub.status.busy":"2022-07-19T12:40:11.433674Z","iopub.execute_input":"2022-07-19T12:40:11.434091Z","iopub.status.idle":"2022-07-19T12:40:11.452453Z","shell.execute_reply.started":"2022-07-19T12:40:11.434054Z","shell.execute_reply":"2022-07-19T12:40:11.451443Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-19T12:40:16.575836Z","iopub.execute_input":"2022-07-19T12:40:16.576270Z","iopub.status.idle":"2022-07-19T12:40:16.606655Z","shell.execute_reply.started":"2022-07-19T12:40:16.576231Z","shell.execute_reply":"2022-07-19T12:40:16.605441Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-19T12:40:24.112175Z","iopub.execute_input":"2022-07-19T12:40:24.112605Z","iopub.status.idle":"2022-07-19T12:40:24.142402Z","shell.execute_reply.started":"2022-07-19T12:40:24.112574Z","shell.execute_reply":"2022-07-19T12:40:24.141143Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Checking the categorical columns","metadata":{}},{"cell_type":"code","source":"categorical_columns_train = []\nfor i in df_train.columns:\n    if df_train[i].dtypes == 'object':\n        categorical_columns_train.append(i)","metadata":{"execution":{"iopub.status.busy":"2022-07-19T12:40:42.357497Z","iopub.execute_input":"2022-07-19T12:40:42.357866Z","iopub.status.idle":"2022-07-19T12:40:42.365058Z","shell.execute_reply.started":"2022-07-19T12:40:42.357837Z","shell.execute_reply":"2022-07-19T12:40:42.363687Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(categorical_columns_train)","metadata":{"execution":{"iopub.status.busy":"2022-07-19T12:40:47.176551Z","iopub.execute_input":"2022-07-19T12:40:47.177945Z","iopub.status.idle":"2022-07-19T12:40:47.184135Z","shell.execute_reply.started":"2022-07-19T12:40:47.177888Z","shell.execute_reply":"2022-07-19T12:40:47.183285Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"categorical_columns_test = []\nfor i in df_test.columns:\n    if df_test[i].dtypes == 'object':\n        categorical_columns_test.append(i)","metadata":{"execution":{"iopub.status.busy":"2022-07-19T12:40:52.426256Z","iopub.execute_input":"2022-07-19T12:40:52.426663Z","iopub.status.idle":"2022-07-19T12:40:52.433670Z","shell.execute_reply.started":"2022-07-19T12:40:52.426629Z","shell.execute_reply":"2022-07-19T12:40:52.432545Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(categorical_columns_test)","metadata":{"execution":{"iopub.status.busy":"2022-07-19T12:40:58.284850Z","iopub.execute_input":"2022-07-19T12:40:58.285286Z","iopub.status.idle":"2022-07-19T12:40:58.292438Z","shell.execute_reply.started":"2022-07-19T12:40:58.285245Z","shell.execute_reply":"2022-07-19T12:40:58.291350Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Handling missing values in categorical columns","metadata":{}},{"cell_type":"code","source":"# checking those category columns which has missing values in train data\n\n\nmiss_catcolumns = []\nfor column in categorical_columns_train:\n    if df_train[column].isnull().sum() > 0:\n        miss_catcolumns.append(column)\n\n\n# checking the no & %age of missing values in column\n\ndef missingCountsInCategoryCol1():\n    missing_col_list = []\n    for i in miss_catcolumns:\n        print(f\"missing vlaues in {i}:- \", df_train[i].isnull().sum())\n        print(f'missing values %age in {i}',df_train[i].isnull().sum() / len(df_train[i]))\n        print('*------------------------------------------------------------*')\n        missing_col_list.append(i)\n    return missing_col_list\n        \nmissingCountsInCategoryCol1()","metadata":{"execution":{"iopub.status.busy":"2022-07-19T12:41:18.997214Z","iopub.execute_input":"2022-07-19T12:41:18.997589Z","iopub.status.idle":"2022-07-19T12:41:19.033035Z","shell.execute_reply.started":"2022-07-19T12:41:18.997558Z","shell.execute_reply":"2022-07-19T12:41:19.032174Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Handling missing Data\n\ndf_train['MasVnrType'] = df_train['MasVnrType'].fillna(df_train['MasVnrType'].mode()[0])\ndf_train['BsmtQual'] = df_train['BsmtQual'].fillna( df_train['BsmtQual'].mode()[0])\ndf_train['BsmtCond'] = df_train['BsmtCond'].fillna( df_train['BsmtCond'].mode()[0])\ndf_train['BsmtExposure'] = df_train['BsmtExposure'].fillna( df_train['BsmtExposure'].mode()[0])\ndf_train['BsmtFinType1'] = df_train['BsmtFinType1'].fillna( df_train['BsmtFinType1'].mode()[0])\ndf_train['BsmtFinType2'] = df_train['BsmtFinType2'].fillna( df_train['BsmtFinType2'].mode()[0])\ndf_train['Electrical'] = df_train['Electrical'].fillna( df_train['Electrical'].mode()[0])\ndf_train['GarageType'] = df_train['GarageType'].fillna( df_train['GarageType'].mode()[0])\ndf_train['GarageFinish'] = df_train['GarageFinish'].fillna( df_train['GarageFinish'].mode()[0])\ndf_train['GarageQual'] = df_train['GarageQual'].fillna( df_train['GarageQual'].mode()[0])\ndf_train['GarageCond'] = df_train['GarageCond'].fillna( df_train['GarageCond'].mode()[0])","metadata":{"execution":{"iopub.status.busy":"2022-07-19T12:41:26.536603Z","iopub.execute_input":"2022-07-19T12:41:26.537614Z","iopub.status.idle":"2022-07-19T12:41:26.563939Z","shell.execute_reply.started":"2022-07-19T12:41:26.537568Z","shell.execute_reply":"2022-07-19T12:41:26.562904Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# checking those category columns which has missing values in test data\n\n\nmiss_catcolumns1 = []\nfor column in categorical_columns_test:\n    if df_test[column].isnull().sum() > 0:\n        miss_catcolumns1.append(column)\n\n\n# checking the no & %age of missing values in column\n\ndef missingCountsInCategoryCol2():\n    missing_col_list1 = []\n    for i in miss_catcolumns1:\n        print(f\"missing vlaues in {i}:- \", df_test[i].isnull().sum())\n        print(f'missing values %age in {i}',df_test[i].isnull().sum() / len(df_test[i]))\n        print('*------------------------------------------------------------*')\n        missing_col_list1.append(i)\n    return missing_col_list1\n        \nmissingColListInTest = missingCountsInCategoryCol2()","metadata":{"execution":{"iopub.status.busy":"2022-07-19T12:41:31.163837Z","iopub.execute_input":"2022-07-19T12:41:31.164942Z","iopub.status.idle":"2022-07-19T12:41:31.204606Z","shell.execute_reply.started":"2022-07-19T12:41:31.164899Z","shell.execute_reply":"2022-07-19T12:41:31.203350Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"missingColListInTest ","metadata":{"execution":{"iopub.status.busy":"2022-07-19T12:41:39.395036Z","iopub.execute_input":"2022-07-19T12:41:39.395438Z","iopub.status.idle":"2022-07-19T12:41:39.403284Z","shell.execute_reply.started":"2022-07-19T12:41:39.395406Z","shell.execute_reply":"2022-07-19T12:41:39.402114Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def handlesMissingColumnsInTest():\n    for i in missingColListInTest:\n        df_test[i] = df_test[i].fillna(df_test[i].mode()[0])\n    return \"Handled\"\n\nhandlesMissingColumnsInTest()","metadata":{"execution":{"iopub.status.busy":"2022-07-19T12:41:45.354360Z","iopub.execute_input":"2022-07-19T12:41:45.354743Z","iopub.status.idle":"2022-07-19T12:41:45.384400Z","shell.execute_reply.started":"2022-07-19T12:41:45.354712Z","shell.execute_reply":"2022-07-19T12:41:45.383268Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-19T12:41:50.726827Z","iopub.execute_input":"2022-07-19T12:41:50.727291Z","iopub.status.idle":"2022-07-19T12:41:50.759484Z","shell.execute_reply.started":"2022-07-19T12:41:50.727253Z","shell.execute_reply":"2022-07-19T12:41:50.758517Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-19T12:41:57.835818Z","iopub.execute_input":"2022-07-19T12:41:57.837231Z","iopub.status.idle":"2022-07-19T12:41:57.864211Z","shell.execute_reply.started":"2022-07-19T12:41:57.837170Z","shell.execute_reply":"2022-07-19T12:41:57.862984Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\ny_train = df_train['SalePrice']\ny_train.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-19T12:42:13.136351Z","iopub.execute_input":"2022-07-19T12:42:13.136737Z","iopub.status.idle":"2022-07-19T12:42:13.143828Z","shell.execute_reply.started":"2022-07-19T12:42:13.136707Z","shell.execute_reply":"2022-07-19T12:42:13.143047Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = df_train.drop(['SalePrice'], axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-07-19T12:42:20.520590Z","iopub.execute_input":"2022-07-19T12:42:20.521522Z","iopub.status.idle":"2022-07-19T12:42:20.529600Z","shell.execute_reply.started":"2022-07-19T12:42:20.521478Z","shell.execute_reply":"2022-07-19T12:42:20.528343Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train_test = df_train.append(df_test)","metadata":{"execution":{"iopub.status.busy":"2022-07-19T12:42:26.389365Z","iopub.execute_input":"2022-07-19T12:42:26.389764Z","iopub.status.idle":"2022-07-19T12:42:26.408435Z","shell.execute_reply.started":"2022-07-19T12:42:26.389733Z","shell.execute_reply":"2022-07-19T12:42:26.407258Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#encoding the categorical column in train data\n\nexclude = ['ExterQual','ExterCond','BsmtQual','BsmtCond','HeatingQC','KitchenQual','GarageQual','GarageCond','BsmtFinType1','BsmtFinType2','BsmtExposure']\nfor i in categorical_columns_train:\n    if i not in exclude:\n        area_dummies = pd.get_dummies(df_train_test[i], prefix=i,drop_first='True')\n        df_train_test = pd.concat([df_train_test, area_dummies],axis=1)\n        df_train_test = df_train_test.drop(columns=[i])","metadata":{"execution":{"iopub.status.busy":"2022-07-19T12:42:32.512188Z","iopub.execute_input":"2022-07-19T12:42:32.512660Z","iopub.status.idle":"2022-07-19T12:42:32.806938Z","shell.execute_reply.started":"2022-07-19T12:42:32.512625Z","shell.execute_reply":"2022-07-19T12:42:32.805604Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train_test.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-19T12:42:37.523290Z","iopub.execute_input":"2022-07-19T12:42:37.523822Z","iopub.status.idle":"2022-07-19T12:42:37.533098Z","shell.execute_reply.started":"2022-07-19T12:42:37.523773Z","shell.execute_reply":"2022-07-19T12:42:37.531498Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#encoding the categorical ordinal column in data\n\ndf_train_test['ExterQual'] = df_train_test['ExterQual'].map({'Ex':1,'Gd':2,'TA':3,'Fa':4,'Po':5})\ndf_train_test['ExterCond'] = df_train_test['ExterCond'].map({'Ex':1,'Gd':2,'TA':3,'Fa':4,'Po':5})\ndf_train_test['BsmtQual'] = df_train_test['BsmtQual'].map({'Ex':1,'Gd':2,'TA':3,'Fa':4,'Po':5})\ndf_train_test['BsmtCond'] = df_train_test['BsmtCond'].map({'Ex':1,'Gd':2,'TA':3,'Fa':4,'Po':5})\ndf_train_test['HeatingQC'] = df_train_test['HeatingQC'].map({'Ex':1,'Gd':2,'TA':3,'Fa':4,'Po':5})\ndf_train_test['KitchenQual'] = df_train_test['KitchenQual'].map({'Ex':1,'Gd':2,'TA':3,'Fa':4,'Po':5})\ndf_train_test['GarageQual'] = df_train_test['GarageQual'].map({'Ex':1,'Gd':2,'TA':3,'Fa':4,'Po':5})\ndf_train_test['GarageCond'] = df_train_test['GarageCond'].map({'Ex':1,'Gd':2,'TA':3,'Fa':4,'Po':5})\ndf_train_test['BsmtExposure'] = df_train_test['BsmtExposure'].map({'Gd':1,'Av':2,'Mn':3,'No':4,})\ndf_train_test['BsmtFinType2'] = df_train_test['BsmtFinType2'].map({'GLQ':1, 'Unf':2, 'ALQ':3,'BLQ':4,'Rec':5,'LwQ':6})\ndf_train_test['BsmtFinType1'] = df_train_test['BsmtFinType1'].map({'GLQ':1, 'Unf':2, 'ALQ':3,'BLQ':4,'Rec':5,'LwQ':6})","metadata":{"execution":{"iopub.status.busy":"2022-07-19T12:42:43.770074Z","iopub.execute_input":"2022-07-19T12:42:43.770488Z","iopub.status.idle":"2022-07-19T12:42:43.801047Z","shell.execute_reply.started":"2022-07-19T12:42:43.770451Z","shell.execute_reply":"2022-07-19T12:42:43.799629Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Splitting of Data\n","metadata":{}},{"cell_type":"code","source":"X_train = df_train_test[:len(df_train)]\nX_test = df_train_test[len(df_train):]","metadata":{"execution":{"iopub.status.busy":"2022-07-19T12:43:05.763905Z","iopub.execute_input":"2022-07-19T12:43:05.764346Z","iopub.status.idle":"2022-07-19T12:43:05.770785Z","shell.execute_reply.started":"2022-07-19T12:43:05.764309Z","shell.execute_reply":"2022-07-19T12:43:05.769636Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_test.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-19T12:43:12.752951Z","iopub.execute_input":"2022-07-19T12:43:12.753434Z","iopub.status.idle":"2022-07-19T12:43:12.761342Z","shell.execute_reply.started":"2022-07-19T12:43:12.753394Z","shell.execute_reply":"2022-07-19T12:43:12.760164Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-19T12:43:17.790000Z","iopub.execute_input":"2022-07-19T12:43:17.790554Z","iopub.status.idle":"2022-07-19T12:43:17.798919Z","shell.execute_reply.started":"2022-07-19T12:43:17.790502Z","shell.execute_reply":"2022-07-19T12:43:17.797779Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_train","metadata":{"execution":{"iopub.status.busy":"2022-07-19T12:43:22.472215Z","iopub.execute_input":"2022-07-19T12:43:22.472575Z","iopub.status.idle":"2022-07-19T12:43:22.482033Z","shell.execute_reply.started":"2022-07-19T12:43:22.472545Z","shell.execute_reply":"2022-07-19T12:43:22.480662Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Applying Linear Regression ","metadata":{}},{"cell_type":"code","source":"from sklearn.linear_model import LinearRegression,Lasso, LassoCV, Ridge,RidgeCV, ElasticNet, ElasticNetCV\nfrom sklearn.preprocessing import StandardScaler, RobustScaler\nfrom sklearn.pipeline import make_pipeline\n\nlinear_reg_pipe = make_pipeline(RobustScaler(), LinearRegression())\nlinear_reg_pipe.fit(X_train,y_train)\nlinear_reg_pipe.score(X_train,y_train)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-19T12:43:38.111135Z","iopub.execute_input":"2022-07-19T12:43:38.111578Z","iopub.status.idle":"2022-07-19T12:43:38.474766Z","shell.execute_reply.started":"2022-07-19T12:43:38.111545Z","shell.execute_reply":"2022-07-19T12:43:38.473313Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Let's create a function to create adjusted R-Squared\ndef adj_r2(x,y):\n    r2 = linear_reg_pipe.score(x,y)\n    n = x.shape[0]\n    p = x.shape[1]\n    adjusted_r2 = 1-(1-r2)*(n-1)/(n-p-1)\n    return adjusted_r2\n\nadj_r2(X_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2022-07-19T12:43:44.354063Z","iopub.execute_input":"2022-07-19T12:43:44.354476Z","iopub.status.idle":"2022-07-19T12:43:44.378701Z","shell.execute_reply.started":"2022-07-19T12:43:44.354440Z","shell.execute_reply":"2022-07-19T12:43:44.377191Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# checking the score of test data\n\ny_pred_linear_reg = linear_reg_pipe.predict(X_test)\nlinear_reg_pipe.fit(X_test, y_pred_linear_reg )\nlinear_reg_pipe.score(X_test, y_pred_linear_reg)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-19T12:43:54.279723Z","iopub.execute_input":"2022-07-19T12:43:54.280537Z","iopub.status.idle":"2022-07-19T12:43:54.438572Z","shell.execute_reply.started":"2022-07-19T12:43:54.280496Z","shell.execute_reply":"2022-07-19T12:43:54.437268Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Lasso Rigression","metadata":{}},{"cell_type":"code","source":"\n\nlasscv = LassoCV(alphas = None,cv =10, max_iter = 100000, normalize = True)\nlasscv.fit(X_train, y_train)\n\nalpha = lasscv.alpha_\nlasso_reg = Lasso(alpha)\nlasso_reg.fit(X_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2022-07-19T12:44:09.596421Z","iopub.execute_input":"2022-07-19T12:44:09.596790Z","iopub.status.idle":"2022-07-19T12:44:12.771625Z","shell.execute_reply.started":"2022-07-19T12:44:09.596761Z","shell.execute_reply":"2022-07-19T12:44:12.770261Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"lasso_reg.score(X_test, y_pred_linear_reg)","metadata":{"execution":{"iopub.status.busy":"2022-07-19T12:44:18.862425Z","iopub.execute_input":"2022-07-19T12:44:18.862787Z","iopub.status.idle":"2022-07-19T12:44:18.877885Z","shell.execute_reply.started":"2022-07-19T12:44:18.862758Z","shell.execute_reply":"2022-07-19T12:44:18.876280Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Ridge Regression","metadata":{}},{"cell_type":"code","source":"# RidgeCV will return best alpha and coefficients after performing 10 cross validations. \n# We will pass an array of random numbers for ridgeCV to select best alpha from them\n\nalphas = np.random.uniform(low=0, high=10, size=(50,))\nridgecv = RidgeCV(alphas = alphas,cv=10,normalize = True)\nridgecv.fit(X_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2022-07-19T12:44:30.694505Z","iopub.execute_input":"2022-07-19T12:44:30.694913Z","iopub.status.idle":"2022-07-19T12:44:55.401180Z","shell.execute_reply.started":"2022-07-19T12:44:30.694878Z","shell.execute_reply":"2022-07-19T12:44:55.400022Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ridge_model = Ridge(alpha=ridgecv.alpha_)\nridge_model.fit(X_train, y_train)\nridge_model.score(X_test, y_pred_linear_reg)","metadata":{"execution":{"iopub.status.busy":"2022-07-19T12:45:25.630133Z","iopub.execute_input":"2022-07-19T12:45:25.630521Z","iopub.status.idle":"2022-07-19T12:45:25.684706Z","shell.execute_reply.started":"2022-07-19T12:45:25.630491Z","shell.execute_reply":"2022-07-19T12:45:25.683372Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"our r2_score for test data (98%) comes same as before using regularization. So, it is fair to say our OLS model didnot overfit the data.","metadata":{}},{"cell_type":"markdown","source":"# Applying RandomForest Regression","metadata":{}},{"cell_type":"code","source":"from sklearn.ensemble import RandomForestRegressor\n\nRandomForest_pipe = make_pipeline(RobustScaler(), RandomForestRegressor())\nRandomForest_pipe.fit(X_train,y_train)\nRandomForest_pipe.score(X_train,y_train)","metadata":{"execution":{"iopub.status.busy":"2022-07-19T12:45:52.938431Z","iopub.execute_input":"2022-07-19T12:45:52.938845Z","iopub.status.idle":"2022-07-19T12:45:56.009737Z","shell.execute_reply.started":"2022-07-19T12:45:52.938809Z","shell.execute_reply":"2022-07-19T12:45:56.008490Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# checking the score of test data\n\ny_pred_RandomForest = RandomForest_pipe.predict(X_test)\nRandomForest_pipe.fit(X_test, y_pred_RandomForest )\nRandomForest_pipe.score(X_test, y_pred_RandomForest)","metadata":{"execution":{"iopub.status.busy":"2022-07-19T12:46:03.051710Z","iopub.execute_input":"2022-07-19T12:46:03.052072Z","iopub.status.idle":"2022-07-19T12:46:05.905391Z","shell.execute_reply.started":"2022-07-19T12:46:03.052043Z","shell.execute_reply":"2022-07-19T12:46:05.904277Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(y_pred_RandomForest)","metadata":{"execution":{"iopub.status.busy":"2022-07-19T12:46:20.480807Z","iopub.execute_input":"2022-07-19T12:46:20.481967Z","iopub.status.idle":"2022-07-19T12:46:20.490175Z","shell.execute_reply.started":"2022-07-19T12:46:20.481911Z","shell.execute_reply":"2022-07-19T12:46:20.488744Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# since we don't need id so removing for train and test data and storing it for submission file\n\nId = df_test['Id'].values\n\ndf_train = df_train.drop(columns=['Id'])\ndf_test = df_test.drop(columns=['Id'])\n","metadata":{"execution":{"iopub.status.busy":"2022-07-19T12:46:25.982566Z","iopub.execute_input":"2022-07-19T12:46:25.983496Z","iopub.status.idle":"2022-07-19T12:46:25.994308Z","shell.execute_reply.started":"2022-07-19T12:46:25.983437Z","shell.execute_reply":"2022-07-19T12:46:25.993169Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"final = pd.DataFrame()\n\n\nfinal['Id'] = Id\nfinal['SalePrice'] = y_pred_RandomForest\n\nfinal","metadata":{"execution":{"iopub.status.busy":"2022-07-19T14:16:08.641617Z","iopub.execute_input":"2022-07-19T14:16:08.642009Z","iopub.status.idle":"2022-07-19T14:16:08.661254Z","shell.execute_reply.started":"2022-07-19T14:16:08.641979Z","shell.execute_reply":"2022-07-19T14:16:08.660318Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### ","metadata":{}}]}