{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# **Importing libraries**","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport matplotlib.pyplot as plt\nimport pandas as pd\nimport seaborn as sns\nsns.set","metadata":{"execution":{"iopub.status.busy":"2022-07-05T15:24:57.800519Z","iopub.execute_input":"2022-07-05T15:24:57.801108Z","iopub.status.idle":"2022-07-05T15:24:58.682246Z","shell.execute_reply.started":"2022-07-05T15:24:57.801014Z","shell.execute_reply":"2022-07-05T15:24:58.681410Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Importing our data set**","metadata":{}},{"cell_type":"code","source":"Train_Data=pd.read_csv('../input/house-prices-advanced-regression-techniques/train.csv')\nTrain_Data.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-05T15:24:58.683941Z","iopub.execute_input":"2022-07-05T15:24:58.684272Z","iopub.status.idle":"2022-07-05T15:24:58.748283Z","shell.execute_reply.started":"2022-07-05T15:24:58.684235Z","shell.execute_reply":"2022-07-05T15:24:58.747027Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Preprocessing**","metadata":{}},{"cell_type":"markdown","source":"## **Exploring The Descriptive Statistics Of The Variables**","metadata":{}},{"cell_type":"code","source":"Train_Data.describe(include=\"all\")","metadata":{"execution":{"iopub.status.busy":"2022-07-05T15:24:58.749429Z","iopub.execute_input":"2022-07-05T15:24:58.749692Z","iopub.status.idle":"2022-07-05T15:24:58.907166Z","shell.execute_reply.started":"2022-07-05T15:24:58.749649Z","shell.execute_reply":"2022-07-05T15:24:58.906505Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## **Determing the variable of interest**","metadata":{}},{"cell_type":"code","source":"#saleprice correlation matrix\ncorrmat = Train_Data.corr()\nk = 10 #number of variables for heatmap\ncols = corrmat.nlargest(k, 'SalePrice')['SalePrice'].index\ncm = np.corrcoef(Train_Data[cols].values.T)\nsns.set(font_scale=1.25)\nplt.subplots(figsize=(20,12))\nhm = sns.heatmap(cm, cbar=True, annot=True, square=True, fmt='.2f', annot_kws={'size': 10}, yticklabels=cols.values, xticklabels=cols.values)\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2022-07-05T15:24:58.909086Z","iopub.execute_input":"2022-07-05T15:24:58.909481Z","iopub.status.idle":"2022-07-05T15:24:59.637923Z","shell.execute_reply.started":"2022-07-05T15:24:58.909444Z","shell.execute_reply":"2022-07-05T15:24:59.637263Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## **Getting The numrical and Categorical Variables for further analysis**","metadata":{}},{"cell_type":"code","source":"numircal_Variables = Train_Data.select_dtypes(include=['int64', 'float64'])\nCategorical_Variables= Train_Data.select_dtypes(include=['object', 'category'])","metadata":{"execution":{"iopub.status.busy":"2022-07-05T15:24:59.638852Z","iopub.execute_input":"2022-07-05T15:24:59.639078Z","iopub.status.idle":"2022-07-05T15:24:59.647421Z","shell.execute_reply.started":"2022-07-05T15:24:59.639045Z","shell.execute_reply":"2022-07-05T15:24:59.646762Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## **Let's Check Numrical variables linearty with sale price**","metadata":{}},{"cell_type":"code","source":"fig = plt.figure(figsize=(30,50))\nplt.subplots_adjust(left=0.1,\n                    bottom=0.1,\n                    right=0.9,\n                    top=1.9,\n                    wspace=0.3,\n                    hspace=0.5)\nfor i, col in enumerate(numircal_Variables.columns):\n        plt.subplot(15,3,i+1)\n        sns.regplot(x=col, y= \"SalePrice\", data=numircal_Variables, scatter_kws={\"color\": \"blue\"}, line_kws={\"color\": \"red\"})\n","metadata":{"execution":{"iopub.status.busy":"2022-07-05T15:24:59.648679Z","iopub.execute_input":"2022-07-05T15:24:59.649078Z","iopub.status.idle":"2022-07-05T15:25:15.939278Z","shell.execute_reply.started":"2022-07-05T15:24:59.649041Z","shell.execute_reply":"2022-07-05T15:25:15.938570Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## **Let's Check our numircal variable**","metadata":{}},{"cell_type":"code","source":"fig_ = numircal_Variables.hist(figsize=(25, 30), bins=50, color=\"darkcyan\",\n                         edgecolor=\"black\", xlabelsize=8, ylabelsize=8)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-05T15:25:15.940442Z","iopub.execute_input":"2022-07-05T15:25:15.940792Z","iopub.status.idle":"2022-07-05T15:25:24.855422Z","shell.execute_reply.started":"2022-07-05T15:25:15.940760Z","shell.execute_reply":"2022-07-05T15:25:24.854763Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## **Let's Plot our Categorical variables**","metadata":{}},{"cell_type":"markdown","source":"first let's add our wanted column to the mix","metadata":{}},{"cell_type":"code","source":"Categorical_Variables['SalePrice']=Train_Data['SalePrice']","metadata":{"execution":{"iopub.status.busy":"2022-07-05T15:25:24.856557Z","iopub.execute_input":"2022-07-05T15:25:24.856945Z","iopub.status.idle":"2022-07-05T15:25:24.863476Z","shell.execute_reply.started":"2022-07-05T15:25:24.856907Z","shell.execute_reply":"2022-07-05T15:25:24.862653Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def boxplot(x, y, **kwargs):\n    sns.boxplot(x=x, y=y)\n    x=plt.xticks(rotation=90)\nf = pd.melt(Train_Data, id_vars=['SalePrice'], value_vars=Categorical_Variables)\ng = sns.FacetGrid(f, col=\"variable\",  col_wrap=2, sharex=False, sharey=False, size=5)\ng = g.map(boxplot, \"value\", \"SalePrice\")\n","metadata":{"execution":{"iopub.status.busy":"2022-07-05T15:25:24.864889Z","iopub.execute_input":"2022-07-05T15:25:24.865244Z","iopub.status.idle":"2022-07-05T15:25:38.331298Z","shell.execute_reply.started":"2022-07-05T15:25:24.865206Z","shell.execute_reply":"2022-07-05T15:25:38.330044Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## **Let's Check For missing Values**","metadata":{}},{"cell_type":"code","source":"pd.set_option('display.max_rows', 500)\nTrain_Data.isnull().sum()  ","metadata":{"execution":{"iopub.status.busy":"2022-07-05T15:25:38.334076Z","iopub.execute_input":"2022-07-05T15:25:38.334423Z","iopub.status.idle":"2022-07-05T15:25:38.353085Z","shell.execute_reply.started":"2022-07-05T15:25:38.334389Z","shell.execute_reply":"2022-07-05T15:25:38.352528Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## **Let's Get the missing values for numircal variables**","metadata":{}},{"cell_type":"code","source":"missing_numircal_variables=numircal_Variables.columns[numircal_Variables .isnull().any()].tolist() \nmissing_numircal_variables","metadata":{"execution":{"iopub.status.busy":"2022-07-05T15:25:38.354223Z","iopub.execute_input":"2022-07-05T15:25:38.354605Z","iopub.status.idle":"2022-07-05T15:25:38.362516Z","shell.execute_reply.started":"2022-07-05T15:25:38.354572Z","shell.execute_reply":"2022-07-05T15:25:38.361682Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## **Encoding all the numircal data with the value with knn**","metadata":{}},{"cell_type":"code","source":"!pip install missingpy","metadata":{"execution":{"iopub.status.busy":"2022-07-05T15:25:38.363763Z","iopub.execute_input":"2022-07-05T15:25:38.364076Z","iopub.status.idle":"2022-07-05T15:25:49.139775Z","shell.execute_reply.started":"2022-07-05T15:25:38.364040Z","shell.execute_reply":"2022-07-05T15:25:49.138930Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.impute import SimpleImputer\nfrom sklearn.impute import KNNImputer\nfrom missingpy import MissForest\nfrom sklearn.experimental import enable_iterative_imputer\nfrom sklearn.impute import IterativeImputer\n\n\n# Impute\nimputer_numircal = IterativeImputer()\nimputer_numircal=imputer_numircal.fit(Train_Data[missing_numircal_variables]) # the upper bound is excluded fit for the indexes 1 and 2\nTrain_Data[missing_numircal_variables]=imputer_numircal.transform(Train_Data[missing_numircal_variables]) # to transform the data\npd.set_option('display.max_rows', 500)\nTrain_Data.isnull().sum()  \n","metadata":{"execution":{"iopub.status.busy":"2022-07-05T15:28:55.174449Z","iopub.execute_input":"2022-07-05T15:28:55.174764Z","iopub.status.idle":"2022-07-05T15:28:55.268313Z","shell.execute_reply.started":"2022-07-05T15:28:55.174728Z","shell.execute_reply":"2022-07-05T15:28:55.267573Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## **Let's Get the missing values for Categorical variables**","metadata":{}},{"cell_type":"code","source":"missing_categorical_variables=Categorical_Variables.columns[Categorical_Variables.isnull().any()].tolist() \nmissing_categorical_variables","metadata":{"execution":{"iopub.status.busy":"2022-07-05T15:29:02.667359Z","iopub.execute_input":"2022-07-05T15:29:02.667814Z","iopub.status.idle":"2022-07-05T15:29:02.690608Z","shell.execute_reply.started":"2022-07-05T15:29:02.667777Z","shell.execute_reply":"2022-07-05T15:29:02.689999Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## **Let's Encode missing Categorical variables with missing**","metadata":{}},{"cell_type":"code","source":"import category_encoders as ce\nimport pandas as pd\nfrom sklearn.impute import SimpleImputer\n# create object of Ordinalencoding\nimputer_categorical= SimpleImputer(missing_values=np.nan,strategy='constant',fill_value=\"missing\") # replace the missing values(NaN) by the mean for every column\nimputer_categorical=imputer_categorical.fit(Train_Data[missing_categorical_variables]) # the upper bound is excluded fit for the indexes 1 and 2\nTrain_Data[missing_categorical_variables]=imputer_categorical.transform(Train_Data[missing_categorical_variables]) # to transform the data\npd.set_option('display.max_rows', 500)\nTrain_Data.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2022-07-05T15:29:02.694407Z","iopub.execute_input":"2022-07-05T15:29:02.696238Z","iopub.status.idle":"2022-07-05T15:29:02.948440Z","shell.execute_reply.started":"2022-07-05T15:29:02.694761Z","shell.execute_reply":"2022-07-05T15:29:02.947479Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Train_Data.select_dtypes(include=np.number).columns.tolist()","metadata":{"execution":{"iopub.status.busy":"2022-07-05T15:29:02.953132Z","iopub.execute_input":"2022-07-05T15:29:02.953516Z","iopub.status.idle":"2022-07-05T15:29:02.973164Z","shell.execute_reply.started":"2022-07-05T15:29:02.953465Z","shell.execute_reply":"2022-07-05T15:29:02.972437Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## **Exploring the pdfs and dealing with outliers**","metadata":{}},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns\n\nfig = plt.subplots(figsize=(12, 36))\ni=0\nfor j, feature in enumerate(numircal_Variables.columns):\n    if feature not in ['Id', 'SalePrice']:\n        i += 1\n        plt.subplot(13, 3, i)\n        sns.histplot(Train_Data[feature], kde=True)\n        plt.tight_layout()\n","metadata":{"execution":{"iopub.status.busy":"2022-07-05T15:29:02.977554Z","iopub.execute_input":"2022-07-05T15:29:02.980569Z","iopub.status.idle":"2022-07-05T15:29:27.776454Z","shell.execute_reply.started":"2022-07-05T15:29:02.980532Z","shell.execute_reply":"2022-07-05T15:29:27.775834Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# checking the stats\nTrain_Data[\"SalePrice\"].describe(include=\"all\")","metadata":{"execution":{"iopub.status.busy":"2022-07-05T15:29:27.778496Z","iopub.execute_input":"2022-07-05T15:29:27.779006Z","iopub.status.idle":"2022-07-05T15:29:27.790263Z","shell.execute_reply.started":"2022-07-05T15:29:27.778932Z","shell.execute_reply":"2022-07-05T15:29:27.789328Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#exploring the pdf\nsns.histplot(Train_Data['SalePrice'])\nplt.tight_layout()\n","metadata":{"execution":{"iopub.status.busy":"2022-07-05T15:29:27.791753Z","iopub.execute_input":"2022-07-05T15:29:27.792213Z","iopub.status.idle":"2022-07-05T15:29:28.195141Z","shell.execute_reply.started":"2022-07-05T15:29:27.792176Z","shell.execute_reply":"2022-07-05T15:29:28.190920Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.catplot(x='SalePrice',kind='box',data=Train_Data)","metadata":{"execution":{"iopub.status.busy":"2022-07-05T15:29:28.196744Z","iopub.execute_input":"2022-07-05T15:29:28.197271Z","iopub.status.idle":"2022-07-05T15:29:28.639408Z","shell.execute_reply.started":"2022-07-05T15:29:28.197226Z","shell.execute_reply":"2022-07-05T15:29:28.638670Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def outliers(df,ft):\n    Q1=df[ft].quantile(0.25)\n    Q3=df[ft].quantile(0.75)\n    IQR=Q3-Q1\n    lower_bound=Q1-1.5*IQR\n    upper_bound=Q3+1.5*IQR\n    lst=df.index[(df[ft]<lower_bound)|(df[ft]>upper_bound)]\n    return lst","metadata":{"execution":{"iopub.status.busy":"2022-07-05T15:29:28.640804Z","iopub.execute_input":"2022-07-05T15:29:28.641227Z","iopub.status.idle":"2022-07-05T15:29:28.647348Z","shell.execute_reply.started":"2022-07-05T15:29:28.641179Z","shell.execute_reply":"2022-07-05T15:29:28.646437Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"index_list=[]\nfor feature in cols: # i chose those columns as there are what we will use for rbm model\n    index_list.extend(outliers(Train_Data,feature))","metadata":{"execution":{"iopub.status.busy":"2022-07-05T15:29:28.648684Z","iopub.execute_input":"2022-07-05T15:29:28.649225Z","iopub.status.idle":"2022-07-05T15:29:28.685278Z","shell.execute_reply.started":"2022-07-05T15:29:28.649186Z","shell.execute_reply":"2022-07-05T15:29:28.684665Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(len(index_list))\nprint(len(Train_Data))","metadata":{"execution":{"iopub.status.busy":"2022-07-05T15:29:28.688711Z","iopub.execute_input":"2022-07-05T15:29:28.690566Z","iopub.status.idle":"2022-07-05T15:29:28.698797Z","shell.execute_reply.started":"2022-07-05T15:29:28.690525Z","shell.execute_reply":"2022-07-05T15:29:28.697922Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def remove(df,ls):\n    ls=sorted(set(ls))\n    df=df.drop(ls)\n    return df","metadata":{"execution":{"iopub.status.busy":"2022-07-05T15:29:28.704926Z","iopub.execute_input":"2022-07-05T15:29:28.706768Z","iopub.status.idle":"2022-07-05T15:29:28.712940Z","shell.execute_reply.started":"2022-07-05T15:29:28.706724Z","shell.execute_reply":"2022-07-05T15:29:28.711974Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#checking stats\nTrain_Data[\"SalePrice\"].describe(include=\"all\")","metadata":{"execution":{"iopub.status.busy":"2022-07-05T15:29:28.714214Z","iopub.execute_input":"2022-07-05T15:29:28.714756Z","iopub.status.idle":"2022-07-05T15:29:28.730174Z","shell.execute_reply.started":"2022-07-05T15:29:28.714717Z","shell.execute_reply":"2022-07-05T15:29:28.729484Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#plot improved\nsns.displot(Train_Data['SalePrice'])","metadata":{"execution":{"iopub.status.busy":"2022-07-05T15:29:28.733799Z","iopub.execute_input":"2022-07-05T15:29:28.736440Z","iopub.status.idle":"2022-07-05T15:29:29.186112Z","shell.execute_reply.started":"2022-07-05T15:29:28.736401Z","shell.execute_reply":"2022-07-05T15:29:29.185412Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Train_Data=remove(Train_Data,index_list)\nTrain_Data_Cleaned=Train_Data.reset_index(drop=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-05T15:29:29.187344Z","iopub.execute_input":"2022-07-05T15:29:29.189787Z","iopub.status.idle":"2022-07-05T15:29:29.196527Z","shell.execute_reply.started":"2022-07-05T15:29:29.189745Z","shell.execute_reply":"2022-07-05T15:29:29.195869Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.displot(Train_Data['SalePrice'])","metadata":{"execution":{"iopub.status.busy":"2022-07-05T15:29:29.197847Z","iopub.execute_input":"2022-07-05T15:29:29.198170Z","iopub.status.idle":"2022-07-05T15:29:29.582108Z","shell.execute_reply.started":"2022-07-05T15:29:29.198133Z","shell.execute_reply":"2022-07-05T15:29:29.581388Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Train_Data_Cleaned.describe(include=\"all\")","metadata":{"execution":{"iopub.status.busy":"2022-07-05T15:29:29.583208Z","iopub.execute_input":"2022-07-05T15:29:29.583883Z","iopub.status.idle":"2022-07-05T15:29:29.733106Z","shell.execute_reply.started":"2022-07-05T15:29:29.583842Z","shell.execute_reply":"2022-07-05T15:29:29.732449Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## **Checking The Ols Assumptions**","metadata":{}},{"cell_type":"code","source":"fig = plt.subplots(figsize=(12, 36))\ni=0\nfor j, feature in enumerate(numircal_Variables.columns):\n    if feature not in ['Id', 'SalePrice']:\n        i += 1\n        plt.subplot(13, 3, i)\n        sns.scatterplot(x=Train_Data_Cleaned[feature], y=Train_Data_Cleaned['SalePrice'])\n        plt.tight_layout()\n","metadata":{"execution":{"iopub.status.busy":"2022-07-05T15:29:29.734581Z","iopub.execute_input":"2022-07-05T15:29:29.734999Z","iopub.status.idle":"2022-07-05T15:29:48.332082Z","shell.execute_reply.started":"2022-07-05T15:29:29.734961Z","shell.execute_reply":"2022-07-05T15:29:48.331458Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#data_cleaned.drop(['SalePrice'],axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-07-05T15:29:48.333614Z","iopub.execute_input":"2022-07-05T15:29:48.334096Z","iopub.status.idle":"2022-07-05T15:29:48.337467Z","shell.execute_reply.started":"2022-07-05T15:29:48.334059Z","shell.execute_reply":"2022-07-05T15:29:48.336833Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Train_Data_Cleaned.columns.values","metadata":{"execution":{"iopub.status.busy":"2022-07-05T15:29:48.338972Z","iopub.execute_input":"2022-07-05T15:29:48.339489Z","iopub.status.idle":"2022-07-05T15:29:48.349796Z","shell.execute_reply.started":"2022-07-05T15:29:48.339455Z","shell.execute_reply":"2022-07-05T15:29:48.348929Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from statsmodels.stats.outliers_influence import variance_inflation_factor\nvariables=Train_Data_Cleaned[['LotArea','LotFrontage']]\nvif=pd.DataFrame()\nvif[\"VIF\"]=[variance_inflation_factor(variables.values,i) for i in range(variables.shape[1])]\nvif['features']=variables.columns\n","metadata":{"execution":{"iopub.status.busy":"2022-07-05T15:29:48.351496Z","iopub.execute_input":"2022-07-05T15:29:48.352200Z","iopub.status.idle":"2022-07-05T15:29:48.367414Z","shell.execute_reply.started":"2022-07-05T15:29:48.352104Z","shell.execute_reply":"2022-07-05T15:29:48.366767Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"vif ","metadata":{"execution":{"iopub.status.busy":"2022-07-05T15:29:48.368456Z","iopub.execute_input":"2022-07-05T15:29:48.368815Z","iopub.status.idle":"2022-07-05T15:29:48.377042Z","shell.execute_reply.started":"2022-07-05T15:29:48.368772Z","shell.execute_reply":"2022-07-05T15:29:48.376206Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Train_Data_Cleaned.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-05T15:29:48.378437Z","iopub.execute_input":"2022-07-05T15:29:48.378886Z","iopub.status.idle":"2022-07-05T15:29:48.405806Z","shell.execute_reply.started":"2022-07-05T15:29:48.378849Z","shell.execute_reply":"2022-07-05T15:29:48.405137Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## **Create Dummies Variables**","metadata":{}},{"cell_type":"code","source":"Train_Data_Cleaned_With_Dummies = pd.get_dummies(Train_Data_Cleaned,drop_first=True)\nCols_Train_Data_Cleaned_With_Dummies= Train_Data_Cleaned_With_Dummies .columns.tolist()\n\nTest_Data=pd.read_csv('../input/house-prices-advanced-regression-techniques/test.csv')\nTest_Data[missing_numircal_variables]=imputer_numircal.transform(Test_Data[missing_numircal_variables]) # to transform the data\nTest_Data[missing_categorical_variables]=imputer_categorical.transform(Test_Data[missing_categorical_variables]) # to transform the data\n\nTest_Data_Cleaned_With_Dummies = pd.get_dummies(Test_Data)\nTest_Data_Cleaned_With_Dummies = Test_Data_Cleaned_With_Dummies.reindex(columns=Cols_Train_Data_Cleaned_With_Dummies).fillna(0)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-05T15:29:48.406921Z","iopub.execute_input":"2022-07-05T15:29:48.407408Z","iopub.status.idle":"2022-07-05T15:29:48.522880Z","shell.execute_reply.started":"2022-07-05T15:29:48.407350Z","shell.execute_reply":"2022-07-05T15:29:48.522077Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Train_Data_Cleaned_With_Dummies.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-05T15:29:48.524157Z","iopub.execute_input":"2022-07-05T15:29:48.524477Z","iopub.status.idle":"2022-07-05T15:29:48.546106Z","shell.execute_reply.started":"2022-07-05T15:29:48.524440Z","shell.execute_reply":"2022-07-05T15:29:48.545476Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Test_Data_Cleaned_With_Dummies.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-05T15:29:48.546996Z","iopub.execute_input":"2022-07-05T15:29:48.547176Z","iopub.status.idle":"2022-07-05T15:29:48.568192Z","shell.execute_reply.started":"2022-07-05T15:29:48.547154Z","shell.execute_reply":"2022-07-05T15:29:48.567514Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **OuR Model**","metadata":{}},{"cell_type":"markdown","source":"## **Declare Inputs and Targets**","metadata":{}},{"cell_type":"code","source":"targets=Train_Data_Cleaned_With_Dummies['SalePrice']\ninputs=Train_Data_Cleaned_With_Dummies.drop(['SalePrice'],axis=1)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-05T15:29:48.569352Z","iopub.execute_input":"2022-07-05T15:29:48.569751Z","iopub.status.idle":"2022-07-05T15:29:48.576634Z","shell.execute_reply.started":"2022-07-05T15:29:48.569717Z","shell.execute_reply":"2022-07-05T15:29:48.575884Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"inputs.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-05T15:29:48.578017Z","iopub.execute_input":"2022-07-05T15:29:48.578840Z","iopub.status.idle":"2022-07-05T15:29:48.598481Z","shell.execute_reply.started":"2022-07-05T15:29:48.578803Z","shell.execute_reply":"2022-07-05T15:29:48.597870Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## **Scaling the Data**","metadata":{}},{"cell_type":"code","source":"from sklearn.preprocessing import StandardScaler\nscaler = StandardScaler()\nscaler.fit(inputs)","metadata":{"execution":{"iopub.status.busy":"2022-07-05T15:29:48.599600Z","iopub.execute_input":"2022-07-05T15:29:48.599981Z","iopub.status.idle":"2022-07-05T15:29:48.619148Z","shell.execute_reply.started":"2022-07-05T15:29:48.599943Z","shell.execute_reply":"2022-07-05T15:29:48.618487Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"inputs_scaled=scaler.transform(inputs)","metadata":{"execution":{"iopub.status.busy":"2022-07-05T15:29:48.622953Z","iopub.execute_input":"2022-07-05T15:29:48.623173Z","iopub.status.idle":"2022-07-05T15:29:48.631901Z","shell.execute_reply.started":"2022-07-05T15:29:48.623147Z","shell.execute_reply":"2022-07-05T15:29:48.631201Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## **Traing and Test Split**","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nx_train, x_test, y_train, y_test = train_test_split(inputs_scaled,targets, test_size = 0.1, random_state = 0)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-05T15:29:48.633282Z","iopub.execute_input":"2022-07-05T15:29:48.633755Z","iopub.status.idle":"2022-07-05T15:29:48.641583Z","shell.execute_reply.started":"2022-07-05T15:29:48.633720Z","shell.execute_reply":"2022-07-05T15:29:48.640775Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## **Create The Model**","metadata":{}},{"cell_type":"code","source":"from sklearn.ensemble import GradientBoostingRegressor\nmodel = GradientBoostingRegressor(n_estimators=3000, learning_rate=0.05,\n                                   max_depth=4, max_features='sqrt',\n                                   min_samples_leaf=15, min_samples_split=10, \n                                   loss='huber', random_state =5)","metadata":{"execution":{"iopub.status.busy":"2022-07-05T15:29:48.644229Z","iopub.execute_input":"2022-07-05T15:29:48.645275Z","iopub.status.idle":"2022-07-05T15:29:48.650145Z","shell.execute_reply.started":"2022-07-05T15:29:48.645238Z","shell.execute_reply":"2022-07-05T15:29:48.649363Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Using K-cross Validation for Better Accuracy**","metadata":{}},{"cell_type":"code","source":"# evaluate model by averaging performance across each fold\nfrom numpy import mean\nfrom numpy import std\nfrom sklearn.datasets import make_blobs\nfrom sklearn.model_selection import KFold\nfrom sklearn.neighbors import KNeighborsClassifier\nfrom sklearn.metrics import accuracy_score\n# create the inputs and outputs\n\nscores = list()\nkfold = KFold(n_splits=10, shuffle=True)\n# enumerate splits\nfor train_ix, test_ix in kfold.split(inputs_scaled):\n    # get data\n    train_X, test_X = inputs_scaled[train_ix], inputs_scaled[test_ix]\n    train_y, test_y = targets[train_ix], targets[test_ix]\n    # fit model\n    model.fit(train_X, train_y)\n    # evaluate model\n    yhat = model.predict(test_X)\n    acc=model.score(test_X,test_y)\n    scores.append(acc)\n    print('> ', acc)\n# summarize model performance\nmean_s, std_s = mean(scores), std(scores)","metadata":{"execution":{"iopub.status.busy":"2022-07-05T15:32:24.231711Z","iopub.execute_input":"2022-07-05T15:32:24.232036Z","iopub.status.idle":"2022-07-05T15:33:54.138535Z","shell.execute_reply.started":"2022-07-05T15:32:24.231996Z","shell.execute_reply":"2022-07-05T15:33:54.137024Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(mean_s,std_s)","metadata":{"execution":{"iopub.status.busy":"2022-07-05T15:33:54.140397Z","iopub.execute_input":"2022-07-05T15:33:54.140672Z","iopub.status.idle":"2022-07-05T15:33:54.145690Z","shell.execute_reply.started":"2022-07-05T15:33:54.140634Z","shell.execute_reply":"2022-07-05T15:33:54.144931Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'''from sklearn.feature_selection import RFE\nfrom sklearn.model_selection import KFold\nfrom sklearn.model_selection import GridSearchCV\n\n# step-1: create a cross-validation scheme\nfolds = KFold(n_splits = 5, shuffle = True, random_state = 100)\nrfe = RFE(model)             \n\n# step-2: specify range of hyperparameters to tune\nhyper_params = [{'n_features_to_select': list(range(1, 14))}]\n# step-3: perform grid search\n# 3.1 specify model\n# 3.2 call GridSearchCV()\nmodel_cv = GridSearchCV(estimator = rfe, \n                        param_grid = hyper_params, \n                        scoring= 'r2', \n                        cv = folds, \n                        verbose = 1,\n                        return_train_score=True)      \n\n# fit the model\nmodel_cv.fit(inputs_scaled,targets)\n'''","metadata":{"execution":{"iopub.status.busy":"2022-07-05T15:33:54.147069Z","iopub.execute_input":"2022-07-05T15:33:54.147571Z","iopub.status.idle":"2022-07-05T15:33:54.158448Z","shell.execute_reply.started":"2022-07-05T15:33:54.147535Z","shell.execute_reply":"2022-07-05T15:33:54.157236Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_hat=model.predict(x_train)","metadata":{"execution":{"iopub.status.busy":"2022-07-05T15:32:05.808227Z","iopub.execute_input":"2022-07-05T15:32:05.808494Z","iopub.status.idle":"2022-07-05T15:32:05.887751Z","shell.execute_reply.started":"2022-07-05T15:32:05.808449Z","shell.execute_reply":"2022-07-05T15:32:05.886938Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.scatter(y_train,y_hat)\nplt.xlabel('Targets (y_train)',size=18)\nplt.ylabel('predictions (y_hat)',size=18)\nplt.show","metadata":{"execution":{"iopub.status.busy":"2022-07-05T15:32:05.889614Z","iopub.execute_input":"2022-07-05T15:32:05.890188Z","iopub.status.idle":"2022-07-05T15:32:06.161159Z","shell.execute_reply.started":"2022-07-05T15:32:05.890140Z","shell.execute_reply":"2022-07-05T15:32:06.160492Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_test_hat=model.predict(x_test)","metadata":{"execution":{"iopub.status.busy":"2022-07-05T15:32:06.162573Z","iopub.execute_input":"2022-07-05T15:32:06.163055Z","iopub.status.idle":"2022-07-05T15:32:06.178486Z","shell.execute_reply.started":"2022-07-05T15:32:06.163016Z","shell.execute_reply":"2022-07-05T15:32:06.177853Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.scatter(y_test,y_test_hat)\nplt.xlabel('Targets (y_test)',size=18)\nplt.ylabel('predictions (y_test_hat)',size=18)\nplt.show","metadata":{"execution":{"iopub.status.busy":"2022-07-05T15:32:06.179755Z","iopub.execute_input":"2022-07-05T15:32:06.180015Z","iopub.status.idle":"2022-07-05T15:32:06.431034Z","shell.execute_reply.started":"2022-07-05T15:32:06.179981Z","shell.execute_reply":"2022-07-05T15:32:06.430348Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Test_Data_Cleaned_With_Dummies=Test_Data_Cleaned_With_Dummies.drop(['SalePrice'],axis=1)\nTest_Data_Cleaned_With_Dummies.head()\nScaled_Test_Data=scaler.transform(Test_Data_Cleaned_With_Dummies)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-05T15:32:06.432406Z","iopub.execute_input":"2022-07-05T15:32:06.432670Z","iopub.status.idle":"2022-07-05T15:32:06.447040Z","shell.execute_reply.started":"2022-07-05T15:32:06.432635Z","shell.execute_reply":"2022-07-05T15:32:06.446260Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_hat_Test_Data=model.predict(Scaled_Test_Data)","metadata":{"execution":{"iopub.status.busy":"2022-07-05T15:32:06.448444Z","iopub.execute_input":"2022-07-05T15:32:06.448715Z","iopub.status.idle":"2022-07-05T15:32:06.542631Z","shell.execute_reply.started":"2022-07-05T15:32:06.448679Z","shell.execute_reply":"2022-07-05T15:32:06.541949Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = pd.DataFrame({'Id':Test_Data['Id'],'SalePrice':y_hat_Test_Data})\nsubmission['SalePrice'] = y_hat_Test_Data\nsubmission.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-05T15:32:06.543700Z","iopub.execute_input":"2022-07-05T15:32:06.543983Z","iopub.status.idle":"2022-07-05T15:32:06.557987Z","shell.execute_reply.started":"2022-07-05T15:32:06.543923Z","shell.execute_reply":"2022-07-05T15:32:06.557306Z"},"trusted":true},"execution_count":null,"outputs":[]}]}