{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"**Inspired by [Soumi Ghosh](https://www.kaggle.com/code/soumighosh99/house-prices-89-score-with-xgboost/notebook), [Srishti Saha](https://www.kaggle.com/code/srishti280992/starter-code-eda-feature-engg-basic-model-v2/notebook?scriptVersionId=6419727), and [Niketan Moon](https://www.kaggle.com/code/niketanmoon/house-prices-prediction-part-2-with-fe/notebook).**","metadata":{}},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebraS\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input/'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-04T03:34:14.867002Z","iopub.execute_input":"2022-08-04T03:34:14.867416Z","iopub.status.idle":"2022-08-04T03:34:14.878208Z","shell.execute_reply.started":"2022-08-04T03:34:14.867383Z","shell.execute_reply":"2022-08-04T03:34:14.876790Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from matplotlib import pyplot as plt\nimport seaborn as sns\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.preprocessing import OrdinalEncoder, OneHotEncoder\nfrom sklearn.model_selection import train_test_split, cross_val_score, GridSearchCV\nfrom xgboost import XGBRegressor\nimport eli5\nfrom eli5.sklearn import PermutationImportance\nfrom sklearn.metrics import mean_squared_error","metadata":{"execution":{"iopub.status.busy":"2022-08-04T03:34:14.939251Z","iopub.execute_input":"2022-08-04T03:34:14.940516Z","iopub.status.idle":"2022-08-04T03:34:14.947854Z","shell.execute_reply.started":"2022-08-04T03:34:14.940463Z","shell.execute_reply":"2022-08-04T03:34:14.946460Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Id is irrelevant until submission.**","metadata":{}},{"cell_type":"code","source":"train_data = pd.read_csv('../input/house-prices-advanced-regression-techniques/train.csv').drop('Id',axis=1)\ntest_data = pd.read_csv('../input/house-prices-advanced-regression-techniques/test.csv')\n# for later use\ntest_ids = test_data.pop('Id')\ntrain_data.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-04T03:34:15.038728Z","iopub.execute_input":"2022-08-04T03:34:15.039566Z","iopub.status.idle":"2022-08-04T03:34:15.108655Z","shell.execute_reply.started":"2022-08-04T03:34:15.039528Z","shell.execute_reply":"2022-08-04T03:34:15.107306Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.describe()","metadata":{"execution":{"iopub.status.busy":"2022-08-04T03:34:15.121831Z","iopub.execute_input":"2022-08-04T03:34:15.122240Z","iopub.status.idle":"2022-08-04T03:34:15.229145Z","shell.execute_reply.started":"2022-08-04T03:34:15.122207Z","shell.execute_reply":"2022-08-04T03:34:15.227918Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data.describe()","metadata":{"execution":{"iopub.status.busy":"2022-08-04T03:34:15.231505Z","iopub.execute_input":"2022-08-04T03:34:15.232175Z","iopub.status.idle":"2022-08-04T03:34:15.340984Z","shell.execute_reply.started":"2022-08-04T03:34:15.232128Z","shell.execute_reply":"2022-08-04T03:34:15.339637Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.info()\ntest_data.info()","metadata":{"execution":{"iopub.status.busy":"2022-08-04T03:34:15.343579Z","iopub.execute_input":"2022-08-04T03:34:15.344221Z","iopub.status.idle":"2022-08-04T03:34:15.381820Z","shell.execute_reply.started":"2022-08-04T03:34:15.344185Z","shell.execute_reply":"2022-08-04T03:34:15.380487Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Some columns use nulls (NaNs) as categories, according to data_description.txt. Let's replace those with 'NA'.**","metadata":{}},{"cell_type":"code","source":"na_not_missing_cols = []\nwith open('../input/house-prices-advanced-regression-techniques/data_description.txt') as file:\n    curr_label = None\n    for line in file:\n        if ':' in line:\n            curr_label = line.split(':')[0]\n        elif 'NA' in line:\n            na_not_missing_cols.append(curr_label)\ntrain_data[na_not_missing_cols].info()\nfor col in na_not_missing_cols:\n    train_data[col].fillna('NA',inplace=True)\n    test_data[col].fillna('NA',inplace=True)\nprint('Above cols in train/test contain nulls after processing: ' + str((train_data[na_not_missing_cols].isnull().any().any())) or (test_data[na_not_missing_cols].isnull().any().any()))","metadata":{"execution":{"iopub.status.busy":"2022-08-04T03:34:15.383512Z","iopub.execute_input":"2022-08-04T03:34:15.383862Z","iopub.status.idle":"2022-08-04T03:34:15.415849Z","shell.execute_reply.started":"2022-08-04T03:34:15.383831Z","shell.execute_reply":"2022-08-04T03:34:15.414609Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Now let's impute for the remaining nulls. We will first add binary columns (new features) to mark nulls in existing labels in train_data (and then do the same for those labels in test_data; note that we cannot add features based on test_data), and then we will use mean for numericals and mode for categoricals. Note that MSSubClass is actually categorical, but it has no nulls (as seen above), and it's already essentially self-ordinal-encoded, so we can just exclude it from this step (putting it through an imputer along with other categorical features could turn it into object dtype, which is unnecessary since it's encoded by default). Likewise, we don't have to worry about nulls in the target variable (if we had nulls in the target, we would probably remove those rows).**","metadata":{}},{"cell_type":"code","source":"# mark nulls\nna_cols = [col for col in train_data.columns if train_data[col].isnull().any()]\n\nnums = list(train_data.select_dtypes(exclude='object').columns)\ncats = list(train_data.select_dtypes(exclude=np.number).columns)\nnums.remove('SalePrice') # target\nnums.remove('MSSubClass') # categorical\n\nimputer = SimpleImputer() # default mean\ntrain_data[nums] = pd.DataFrame(imputer.fit_transform(train_data[nums]),columns=nums)\ntest_data[nums] = pd.DataFrame(imputer.transform(test_data[nums]),columns=nums) # reuse mean from training\nimputer = SimpleImputer(strategy='most_frequent')\ntrain_data[cats] = pd.DataFrame(imputer.fit_transform(train_data[cats]),columns=cats)\ntest_data[cats] = pd.DataFrame(imputer.transform(test_data[cats]),columns=cats) # reuse mode from training\nfor col in na_cols: # mark nulls (binary-encoded categorical features)\n    train_data['na_'+col] = train_data[col].isnull().astype('int64')\n    test_data['na_'+col] = test_data[col].isnull().astype('int64')\nprint('Any nulls in train/test after processing: ' + str((train_data.isnull().any().any())) or (test_data.isnull().any().any()))","metadata":{"execution":{"iopub.status.busy":"2022-08-04T03:34:15.450773Z","iopub.execute_input":"2022-08-04T03:34:15.451765Z","iopub.status.idle":"2022-08-04T03:34:15.556678Z","shell.execute_reply.started":"2022-08-04T03:34:15.451724Z","shell.execute_reply":"2022-08-04T03:34:15.555462Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Next comes encoding the categorical variables. We will apply ordinal encoding to some features and dummy/one-hot encoding to others. Note that we don't care about multicollinearity since decision trees are immune to it.**","metadata":{}},{"cell_type":"code","source":"search = ['Street',\n          'LotShape',\n          'Utilities',\n          'LandSlope',\n          'BsmtFinType1',\n          'BsmtFinType2',\n          'BsmtExposure',\n          'CentralAir',\n          'Functional'\n          'PavedDrive'] # looked at data_description.txt to determine which were obviously ordinal\nordinal_cols = []\nall_categories = []\nwith open('../input/house-prices-advanced-regression-techniques/data_description.txt') as file:\n    searching = False\n    curr_label = None\n    curr_categories = []\n    for line in file:\n        line = line.strip('\\n').strip()\n        if line == '':\n            if curr_categories: # empty string and non-empty list indicates end of categories\n                ordinal_cols.append(curr_label)\n                all_categories.append(curr_categories)\n                searching = False\n                curr_label = None\n                curr_categories = []\n        elif ':' in line: # checking for ordinal categorical variables\n            curr_label = line.split(':')[0]\n            if 'Overall' in curr_label:\n                pass\n            elif 'Q' in curr_label:\n                searching = True\n            elif curr_label[-4:] == 'Cond':\n                searching = True\n            else:\n                for string in search:\n                    if string in curr_label:\n                        search.remove(string)\n                        searching = True\n            if not searching:\n                curr_label = None\n        elif searching: # not an empty string but searching means currently on a category\n            split = line.split()\n            if len(split)>1:\n                curr_categories.append(split[0])\n                \nfor i,label in enumerate(ordinal_cols):\n    print(label + '-'*(15-len(label)) + str(all_categories[i]))\n    cats.remove(label)\n    \nencoder = OrdinalEncoder(categories=all_categories) # thankfully no errors caused by unknown values (typos)\n                \ntrain_data[ordinal_cols] = encoder.fit_transform(train_data[ordinal_cols])\ntest_data[ordinal_cols] = encoder.transform(test_data[ordinal_cols])\n\ntrain_data[ordinal_cols].head()","metadata":{"execution":{"iopub.status.busy":"2022-08-04T03:34:15.558799Z","iopub.execute_input":"2022-08-04T03:34:15.559172Z","iopub.status.idle":"2022-08-04T03:34:15.647727Z","shell.execute_reply.started":"2022-08-04T03:34:15.559138Z","shell.execute_reply":"2022-08-04T03:34:15.646416Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(train_data[cats].nunique()) # none of them need target encoding\n# graphing to see if there is anything worth dropping\ndef bar_plot(col):\n    grouped = train_data.SalePrice.groupby(train_data[col]).mean().reset_index()\n    sns.barplot(x=grouped[col],y=grouped.SalePrice,order=grouped.sort_values('SalePrice')[col])\n    plt.title(col+' vs. SalePrice')\n    plt.xticks(rotation=60)\n\nplt.figure(figsize=(20,40))\nplt.subplots_adjust(wspace=0.5,hspace=1)\ni=1\nfor col in cats:\n    plt.subplot(9,3,i)\n    bar_plot(col)\n    i+=1\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-04T03:34:15.649698Z","iopub.execute_input":"2022-08-04T03:34:15.650377Z","iopub.status.idle":"2022-08-04T03:34:20.081496Z","shell.execute_reply.started":"2022-08-04T03:34:15.650305Z","shell.execute_reply":"2022-08-04T03:34:20.080095Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**I see something weird! 'C (all)' is used instead of 'C', and upon further investigation, I found even more inconsistencies with the dataset's documentation (data_description.txt), including typos. Let's fix that.**","metadata":{}},{"cell_type":"code","source":"labels = train_data.columns\nall_categories = []\nwith open('../input/house-prices-advanced-regression-techniques/data_description.txt') as file:\n    searching = False\n    curr_label = None\n    curr_categories = []\n    for line in file:\n        line = line.strip('\\n').strip()\n        if line == '':\n            if curr_categories: # empty string and non-empty list indicates end of categories\n                all_categories.append(curr_categories)\n                searching = False\n                curr_label = None\n                curr_categories = []\n        elif ':' in line and not searching: # checking for categorical variables\n            curr_label = line.split(':')[0]\n            if curr_label in cats:\n                searching = True\n            if not searching:\n                curr_label = None\n        elif searching: # not an empty string but searching means currently on a category\n            split = line.split()\n            if len(split)>1:\n                curr_categories.append(split[0].lower())\n# last variable has no new line afterward\nall_categories.append(curr_categories)\n# change wd to wd sdng\nall_categories[cats.index('Exterior1st')][all_categories[cats.index('Exterior1st')].index('wd')] = 'wd sdng'\nall_categories[cats.index('Exterior2nd')][all_categories[cats.index('Exterior2nd')].index('wd')] = 'wd sdng'\n# handling typos\ntrain_data[cats] = train_data[cats].applymap(lambda x:x.lower())\ntest_data[cats] = test_data[cats].applymap(lambda x:x.lower())\n\ndef fix_typo1(value):\n    if value == 'brk cmn':\n        return 'brkcomm'\n    elif value == 'cmentbd':\n        return 'cemntbd'\n    elif value == 'wd shng':\n        return 'wdshing'\n    return value\n\ndef fix_typo2(value):\n    if value == 'duplex':\n        return 'duplx'\n    elif value == 'twnhs':\n        return 'twnhsi'\n    return value\n\ndef fix_typo3(value):\n    if value == 'c (all)':\n        return 'c'\n    return value\n\ntrain_data.Exterior2nd = train_data.Exterior2nd.apply(fix_typo1)\ntest_data.Exterior2nd = test_data.Exterior2nd.apply(fix_typo1)\ntrain_data.BldgType = train_data.BldgType.apply(fix_typo2)\ntest_data.BldgType = test_data.BldgType.apply(fix_typo2)\ntrain_data.MSZoning = train_data.MSZoning.apply(fix_typo3)\ntest_data.MSZoning = test_data.MSZoning.apply(fix_typo3)\n\n# code used to check for typos\n\ntypos = {}\nfor i,cat in enumerate(cats):\n    for j,value in train_data[cat].items():\n        if value not in all_categories[i]:\n            if typos.get(value) is None:\n                typos[value] = 1\n            else:\n                typos[value] += 1\n            print(cat + \" : \" + str(j) + ' : ' + value + ' : ' + str(all_categories[i]))\nif typos:\n    print(typos)\nelse:\n    print('No more typos in train_data.')\n            \ntypos = {}\nfor i,cat in enumerate(cats):\n    for j,value in test_data[cat].items():\n        if value not in all_categories[i]:\n            if typos.get(value) is None:\n                typos[value] = 1\n            else:\n                typos[value] += 1\n            print(cat + \" : \" + str(j) + ' : ' + value + ' : ' + str(all_categories[i]))\nif typos:\n    print(typos)\nelse:\n    print('No more typos in test_data.')\n\n# wd sdng has space in it (Exterior1st & Exterior2nd)\n# duplex -> duplx (BldgType)\n# twnhs -> twnhsi (BldgType)\n# cmentbd -> cemntbd (Exterior2nd)\n# wd shng -> wdshing (Exterior2nd)\n# brk cmn -> brkcomm (Exterior2nd)\n\n# correct = {}\n# for i,cat in enumerate(cats):\n#     for j,value in train_data[cat].items():\n#         if value in all_categories[i]:\n#             if correct.get(value) is None:\n#                 correct[value] = 1\n#             else:\n#                 correct[value] += 1\n# # print(correct['duplx']) # error, see below\n# print(correct['twnhse'])\n# # print(correct['twnhsi']) # error, see below\n# print(correct['cemntbd'])\n# print(correct['wdshing'])\n# print(correct['brkcomm'])\n# print(correct['brkcmn'])\n\n# correct = {}\n# for i,cat in enumerate(cats):\n#     for j,value in test_data[cat].items():\n#         if value in all_categories[i]:\n#             if correct.get(value) is None:\n#                 correct[value] = 1\n#             else:\n#                 correct[value] += 1\n# # print(correct['duplx'])\n# print(correct['twnhse'])\n# # print(correct['twnhsi']) # duplex is used instead of duplx, and twnhs is used instead of twnhsi\n# print(correct['cemntbd'])\n# print(correct['wdshing'])\n# print(correct['brkcomm'])\n# print(correct['brkcmn'])","metadata":{"execution":{"iopub.status.busy":"2022-08-04T03:34:20.084012Z","iopub.execute_input":"2022-08-04T03:34:20.084507Z","iopub.status.idle":"2022-08-04T03:34:20.193256Z","shell.execute_reply.started":"2022-08-04T03:34:20.084457Z","shell.execute_reply":"2022-08-04T03:34:20.191952Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i,label in enumerate(cats):\n    print(label + '-'*(15-len(label)) + str(all_categories[i]))\n\nonehot = OneHotEncoder(categories=all_categories,sparse=False) # categories not in training set are ignored, return NumPy array\noh_train = pd.DataFrame(onehot.fit_transform(train_data[cats])) # FutureWarning isn't our problem\noh_test = pd.DataFrame(onehot.transform(test_data[cats]))\noh_train.columns = onehot.get_feature_names_out()\noh_test.columns = onehot.get_feature_names_out() # add back column labels\ntrain_data = pd.concat([train_data.drop(columns=cats),oh_train],axis=1)\ntest_data = pd.concat([test_data.drop(columns=cats),oh_test],axis=1)\n\nprint('train_data and test_data have same columns (except SalePrice): ' + str(sum(train_data.drop(columns='SalePrice').columns==test_data.columns)==len(train_data.drop(columns='SalePrice').columns)==len(test_data.columns)))","metadata":{"execution":{"iopub.status.busy":"2022-08-04T03:34:20.194860Z","iopub.execute_input":"2022-08-04T03:34:20.195311Z","iopub.status.idle":"2022-08-04T03:34:20.262834Z","shell.execute_reply.started":"2022-08-04T03:34:20.195269Z","shell.execute_reply":"2022-08-04T03:34:20.261295Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Now we need to process our numerical variables, including the target. Note that feature scaling is redundant with decision trees, and decision trees are robust to outliers in the features, but transforming the target and dealing with its outliers can help.**","metadata":{}},{"cell_type":"code","source":"sns.histplot(train_data.SalePrice)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T03:34:20.266178Z","iopub.execute_input":"2022-08-04T03:34:20.266688Z","iopub.status.idle":"2022-08-04T03:34:20.576618Z","shell.execute_reply.started":"2022-08-04T03:34:20.266638Z","shell.execute_reply":"2022-08-04T03:34:20.575437Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def cap_outliers(col):\n    q1 = col.describe()[4]\n    q3 = col.describe()[6]\n    margin = 1.5*(q3-q1)\n    upper = q3 + margin\n    lower = q1 - margin\n    col.mask(col>upper,upper,inplace=True)\n    col.mask(col<lower,lower,inplace=True)\n\ntrain_data.SalePrice = np.log(train_data.SalePrice)\ncap_outliers(train_data.SalePrice)\nsns.histplot(train_data.SalePrice)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T03:34:20.578570Z","iopub.execute_input":"2022-08-04T03:34:20.579335Z","iopub.status.idle":"2022-08-04T03:34:20.894218Z","shell.execute_reply.started":"2022-08-04T03:34:20.579270Z","shell.execute_reply":"2022-08-04T03:34:20.892958Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Let's use permutation importance to help us select features.**","metadata":{}},{"cell_type":"code","source":"y = train_data.SalePrice\nX = train_data.drop(columns='SalePrice')\nxgb_model = XGBRegressor(n_estimators=950,learning_rate=0.05,n_jobs=-1)\nprint(f'cross_val_score (r2) = {np.mean(cross_val_score(xgb_model,X,y,n_jobs=-1))}')","metadata":{"execution":{"iopub.status.busy":"2022-08-04T03:34:20.895750Z","iopub.execute_input":"2022-08-04T03:34:20.896129Z","iopub.status.idle":"2022-08-04T03:35:08.126275Z","shell.execute_reply.started":"2022-08-04T03:34:20.896084Z","shell.execute_reply":"2022-08-04T03:35:08.125064Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train,X_valid,y_train,y_valid = train_test_split(X,y,random_state=0,test_size=0.2)\nxgb_model.fit(X_train,y_train,verbose=False)\nperm = PermutationImportance(xgb_model,random_state=0,n_iter=10).fit(X_valid,y_valid) # n_iter makes this slow but precise\nimportances = eli5.explain_weights_df(perm,feature_names=list(X.columns)).sort_values(by='weight',ascending=False)\nimportances.index = importances.feature\nimportances.drop(columns='feature',inplace=True)\nprint(importances.to_string())","metadata":{"execution":{"iopub.status.busy":"2022-08-04T03:35:08.128049Z","iopub.execute_input":"2022-08-04T03:35:08.128960Z","iopub.status.idle":"2022-08-04T03:36:00.840667Z","shell.execute_reply.started":"2022-08-04T03:35:08.128868Z","shell.execute_reply":"2022-08-04T03:36:00.839246Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# for i in np.arange(3,9): # choosing a cutoff\n#     print(f'cutoff = {i}')\n#     chosen = list(importances[importances.weight>=1/(10**i)].index)\n#     print(f'{len(chosen)} features selected out of {len(importances.index)}')\n#     X = train_data[list(chosen)]\n#     xgb_model = XGBRegressor(n_estimators=950,learning_rate=0.05,n_jobs=-1)\n#     print(f'cross_val_score = {np.mean(cross_val_score(xgb_model,X,y,n_jobs=-1))}')\n#     print()\n\nchosen = list(importances[importances.weight>=0.000001].index)\nfor feature in chosen:\n    print(feature)\nprint(f'{len(chosen)} features selected out of {len(importances.index)}')\ntrain_data = train_data[['SalePrice']+list(chosen)]\ntest_data = test_data[chosen]\nX = train_data.drop(columns='SalePrice')\nxgb_model = XGBRegressor(n_estimators=950,learning_rate=0.05,n_jobs=-1)\nprint(f'cross_val_score = {np.mean(cross_val_score(xgb_model,X,y,n_jobs=-1))}')","metadata":{"execution":{"iopub.status.busy":"2022-08-04T03:36:00.842444Z","iopub.execute_input":"2022-08-04T03:36:00.843221Z","iopub.status.idle":"2022-08-04T03:36:29.102192Z","shell.execute_reply.started":"2022-08-04T03:36:00.843170Z","shell.execute_reply":"2022-08-04T03:36:29.101281Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Time for feature engineering! Decision trees aren't great at learning counts (they cannot easily sum information from many features at once).**","metadata":{}},{"cell_type":"code","source":"non_dummy_cols = [col for col in list(train_data.columns) if '_' not in col]\ncorr_mat = abs(train_data[non_dummy_cols].corr())\nfig,ax = plt.subplots(figsize=(40,30))\nsns.set(font_scale=1.05)\nsns.heatmap(corr_mat,annot=True,mask=np.triu(corr_mat))\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-04T03:36:29.103207Z","iopub.execute_input":"2022-08-04T03:36:29.104644Z","iopub.status.idle":"2022-08-04T03:36:34.344286Z","shell.execute_reply.started":"2022-08-04T03:36:29.104600Z","shell.execute_reply":"2022-08-04T03:36:34.343026Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"to_add = ['GrLivArea','TotalBsmtSF','LotArea','GarageArea','WoodDeckSF','ScreenPorch','PoolArea','MasVnrArea','OpenPorchSF','3SsnPorch']\n# none: 0.918705749881274\n# all: 0.9187827357696641\n# testing removal of one:\n# - GrLivArea: 0.9188670726085466\n# - TotalBsmtSF: 0.9187886314666489\n# - LotArea: 0.9197009218268437\n# - LotFrontage: 0.9204940516208472 <<<\n# - GarageArea: 0.919545455542074\n# - WoodDeckSF: 0.9191879655838413\n# - ScreenPorch: 0.9189140539707881\n# - PoolArea: 0.9184020036506262\n# - MasVnrArea: 0.9178427231901451\n# - OpenPorchSF: 0.9183481807614333\n# - 3SsnPorch: 0.9184157248284194\ntrain_data['SumArea'] = train_data.loc[:,to_add].sum(axis=1)\ntest_data['SumArea'] = test_data.loc[:,to_add].sum(axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T03:36:34.347644Z","iopub.execute_input":"2022-08-04T03:36:34.348004Z","iopub.status.idle":"2022-08-04T03:36:34.360850Z","shell.execute_reply.started":"2022-08-04T03:36:34.347972Z","shell.execute_reply":"2022-08-04T03:36:34.359373Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Almost done! Next comes tuning hyperparameters (commented out because it takes several hours).**","metadata":{}},{"cell_type":"code","source":"X = train_data.drop(columns='SalePrice')\nX_train,X_valid,y_train,y_valid = train_test_split(X,y,random_state=0,test_size=0.2)\n\n# params = {'max_depth':np.arange(3,10,2),\n#           'min_child_weight':np.arange(1,6,2),\n#           'gamma':[i/10 for i in np.arange(0,4)],\n#           'subsample':[i/10 for i in np.arange(7,10)],\n#           'alpha':[0,0.001,0.005,0.01,0.05]}\n# grid = GridSearchCV(xgb_model,param_grid=params,scoring='r2',return_train_score=True,verbose=2,n_jobs=-1)\n# grid.fit(X_train,y_train)\n# print(f'best_params_ = {grid.best_params_}')\n\n# best_params_ = best_params_ = {'alpha': 0, 'gamma': 0.0, 'max_depth': 3, 'min_child_weight': 3, 'subsample': 0.9}","metadata":{"execution":{"iopub.status.busy":"2022-08-04T03:37:02.775178Z","iopub.execute_input":"2022-08-04T03:37:02.775591Z","iopub.status.idle":"2022-08-04T03:37:02.785236Z","shell.execute_reply.started":"2022-08-04T03:37:02.775532Z","shell.execute_reply":"2022-08-04T03:37:02.783980Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Let's take a deeper look.**","metadata":{}},{"cell_type":"code","source":"# xgb_model = XGBRegressor(n_estimators=950,learning_rate=0.05,n_jobs=-1,max_depth=3,min_child_weight=3,subsample=0.9)\n# print(f'cross_val_score (for subsample=0.9) = {np.mean(cross_val_score(xgb_model,X,y,n_jobs=-1))}')\n\nxgb_model = XGBRegressor(n_estimators=950,learning_rate=0.05,n_jobs=-1,max_depth=3,min_child_weight=3)\nprint(f'cross_val_score (for default subsample=1) = {np.mean(cross_val_score(xgb_model,X,y,n_jobs=-1))}')\n\n# default subsample is better\n\n\n\n# xgb_model = XGBRegressor(n_estimators=950,learning_rate=0.05,n_jobs=-1,max_depth=2,min_child_weight=3)\n# print(f'cross_val_score (for max_depth=2) = {np.mean(cross_val_score(xgb_model,X,y,n_jobs=-1))}')\n\n# xgb_model = XGBRegressor(n_estimators=950,learning_rate=0.05,n_jobs=-1,max_depth=4,min_child_weight=3)\n# print(f'cross_val_score (for max_depth=4) = {np.mean(cross_val_score(xgb_model,X,y,n_jobs=-1))}')\n\n# max_depth=3 is still best\n\n\n\nxgb_model = XGBRegressor(n_estimators=950,learning_rate=0.05,n_jobs=-1,max_depth=3,min_child_weight=2)\nprint(f'cross_val_score (for min_child_weight=2) = {np.mean(cross_val_score(xgb_model,X,y,n_jobs=-1))}')\n\n# min_child_weight=2 is better","metadata":{"execution":{"iopub.status.busy":"2022-08-04T03:37:02.786571Z","iopub.execute_input":"2022-08-04T03:37:02.787708Z","iopub.status.idle":"2022-08-04T03:37:38.384217Z","shell.execute_reply.started":"2022-08-04T03:37:02.787660Z","shell.execute_reply":"2022-08-04T03:37:38.383043Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Let's submit! Note that I have to use exp to undo the log transformation of the predictions.**","metadata":{}},{"cell_type":"code","source":"xgb_model.fit(X_train,y_train,verbose=False)\nprint(f'RMSLE = {mean_squared_error(np.log1p(np.exp(y_valid)),np.log1p(np.exp(xgb_model.predict(X_valid))),squared=False)}') # evaluate on validation set using same metric used in Kaggle competition, for comparison\npd.DataFrame({'Id':test_ids,'SalePrice':np.exp(xgb_model.predict(test_data))}).to_csv('submission.csv',index=False)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T03:37:38.386026Z","iopub.execute_input":"2022-08-04T03:37:38.386393Z","iopub.status.idle":"2022-08-04T03:37:42.993897Z","shell.execute_reply.started":"2022-08-04T03:37:38.386360Z","shell.execute_reply":"2022-08-04T03:37:42.992454Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Possible improvements include:**\n* **stacking**","metadata":{}}]}