{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-09T12:59:52.432718Z","iopub.execute_input":"2022-07-09T12:59:52.433231Z","iopub.status.idle":"2022-07-09T12:59:52.462912Z","shell.execute_reply.started":"2022-07-09T12:59:52.433143Z","shell.execute_reply":"2022-07-09T12:59:52.461720Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom sklearn import set_config; set_config(display='diagram')\nfrom sklearn.pipeline import Pipeline, make_pipeline\nfrom sklearn.preprocessing import MinMaxScaler, OneHotEncoder, OrdinalEncoder\nfrom sklearn.compose import ColumnTransformer, make_column_transformer\nfrom sklearn.impute import SimpleImputer, KNNImputer\nfrom sklearn.tree import DecisionTreeRegressor\nfrom sklearn.metrics import make_scorer\nfrom sklearn.model_selection import cross_val_score, GridSearchCV, RandomizedSearchCV, StratifiedKFold\nfrom sklearn.feature_selection import SelectPercentile, chi2\nfrom sklearn.linear_model import LinearRegression, Ridge\nfrom sklearn.ensemble import RandomForestRegressor\nfrom xgboost import XGBRegressor","metadata":{"execution":{"iopub.status.busy":"2022-07-09T13:49:12.412157Z","iopub.execute_input":"2022-07-09T13:49:12.412571Z","iopub.status.idle":"2022-07-09T13:49:12.420255Z","shell.execute_reply.started":"2022-07-09T13:49:12.412539Z","shell.execute_reply":"2022-07-09T13:49:12.419005Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv(\"/kaggle/input/house-prices-advanced-regression-techniques/train.csv\")\ndf_test = pd.read_csv(\"/kaggle/input/house-prices-advanced-regression-techniques/test.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-07-09T13:01:08.775911Z","iopub.execute_input":"2022-07-09T13:01:08.777143Z","iopub.status.idle":"2022-07-09T13:01:08.851429Z","shell.execute_reply.started":"2022-07-09T13:01:08.777091Z","shell.execute_reply":"2022-07-09T13:01:08.849970Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 1. BASELINE","metadata":{}},{"cell_type":"markdown","source":"### 1.1 Initial feature overview","metadata":{}},{"cell_type":"code","source":"df.dtypes.value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-07-09T13:03:41.212058Z","iopub.execute_input":"2022-07-09T13:03:41.212437Z","iopub.status.idle":"2022-07-09T13:03:41.228050Z","shell.execute_reply.started":"2022-07-09T13:03:41.212405Z","shell.execute_reply":"2022-07-09T13:03:41.226851Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"feat_categorical_nunique = df.select_dtypes(include=['object'])\nfeat_categorical_nunique = feat_categorical_nunique.nunique()\nfeat_categorical_nunique.sum()","metadata":{"execution":{"iopub.status.busy":"2022-07-09T13:04:41.299506Z","iopub.execute_input":"2022-07-09T13:04:41.299946Z","iopub.status.idle":"2022-07-09T13:04:41.321850Z","shell.execute_reply.started":"2022-07-09T13:04:41.299907Z","shell.execute_reply":"2022-07-09T13:04:41.320511Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.histplot(data=feat_categorical_nunique);","metadata":{"execution":{"iopub.status.busy":"2022-07-09T13:04:43.625782Z","iopub.execute_input":"2022-07-09T13:04:43.626534Z","iopub.status.idle":"2022-07-09T13:04:43.905565Z","shell.execute_reply.started":"2022-07-09T13:04:43.626489Z","shell.execute_reply":"2022-07-09T13:04:43.904053Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# store only 7 categorigal features\nfeat_categorical_small = feat_categorical_nunique[feat_categorical_nunique < 7]","metadata":{"execution":{"iopub.status.busy":"2022-07-09T13:05:14.687305Z","iopub.execute_input":"2022-07-09T13:05:14.687704Z","iopub.status.idle":"2022-07-09T13:05:14.694382Z","shell.execute_reply.started":"2022-07-09T13:05:14.687673Z","shell.execute_reply":"2022-07-09T13:05:14.692619Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 1.2 Baseline pipeline","metadata":{}},{"cell_type":"markdown","source":"#### a) Preprocess","metadata":{}},{"cell_type":"code","source":"categorical = list(df.select_dtypes(include=['object']).loc[:, df.nunique() < 7])\nnumerical = list(df.select_dtypes(exclude=['object']).drop(columns=['Id', 'SalePrice']))\n\n# cat pipeline\ncat_pipe = Pipeline([\n    ('imputer', SimpleImputer(strategy='most_frequent')),\n    ('encoder', OneHotEncoder(handle_unknown='ignore', sparse=True))\n])\n\n# num pipeline\nnum_pipe = Pipeline([\n    ('imputer', SimpleImputer(strategy='mean')),\n    ('scaler', MinMaxScaler())\n])\n\npreprocessor = ColumnTransformer([\n    ('cat', cat_pipe, categorical),\n    ('num', num_pipe, numerical)\n])\n\npreproc_baseline = Pipeline([\n    ('preprocessor', preprocessor)\n])\n\nshape_preproc_baseline = preproc_baseline.fit_transform(df)\n\n# create DataFrame\nshape_preproc_baseline = pd.DataFrame(data=shape_preproc_baseline).shape","metadata":{"execution":{"iopub.status.busy":"2022-07-09T13:06:26.812256Z","iopub.execute_input":"2022-07-09T13:06:26.812674Z","iopub.status.idle":"2022-07-09T13:06:26.881602Z","shell.execute_reply.started":"2022-07-09T13:06:26.812640Z","shell.execute_reply":"2022-07-09T13:06:26.880302Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### b) Add estimator","metadata":{}},{"cell_type":"code","source":"pipe_baseline = Pipeline([\n                        ('preproc', preproc_baseline),\n                        ('model', DecisionTreeRegressor())\n                        ])","metadata":{"execution":{"iopub.status.busy":"2022-07-09T13:07:55.078800Z","iopub.execute_input":"2022-07-09T13:07:55.079347Z","iopub.status.idle":"2022-07-09T13:07:55.085533Z","shell.execute_reply.started":"2022-07-09T13:07:55.079288Z","shell.execute_reply":"2022-07-09T13:07:55.084620Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### c) Cross-Validate","metadata":{}},{"cell_type":"code","source":"X = df.drop(columns=['SalePrice', 'Id'])\ny = df['SalePrice']\n\nscore_baseline = cross_val_score(pipe_baseline, X, y, cv=5, scoring='r2', n_jobs=-1).mean()\nscore_baseline","metadata":{"execution":{"iopub.status.busy":"2022-07-09T13:12:11.980907Z","iopub.execute_input":"2022-07-09T13:12:11.981295Z","iopub.status.idle":"2022-07-09T13:12:12.221564Z","shell.execute_reply.started":"2022-07-09T13:12:11.981263Z","shell.execute_reply":"2022-07-09T13:12:12.220802Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 2. XGBoost","metadata":{}},{"cell_type":"markdown","source":"### 2.1 Create Pipeline","metadata":{}},{"cell_type":"code","source":"# num pipeline\nnums = sorted(X.select_dtypes(include=[\"int64\", \"float64\"]).columns)\n\nnums_pipe = Pipeline([\n    ('imputer', KNNImputer()),\n    ('scaler', MinMaxScaler())\n])","metadata":{"execution":{"iopub.status.busy":"2022-07-09T13:39:29.105255Z","iopub.execute_input":"2022-07-09T13:39:29.105628Z","iopub.status.idle":"2022-07-09T13:39:29.113057Z","shell.execute_reply.started":"2022-07-09T13:39:29.105600Z","shell.execute_reply":"2022-07-09T13:39:29.112102Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ord pipeline\n\nfeat_ordinal_dict = {\n    # considers \"missing\" as \"neutral\"\n    \"BsmtCond\": ['missing', 'Po', 'Fa', 'TA', 'Gd'],\n    \"BsmtExposure\": ['missing', 'No', 'Mn', 'Av', 'Gd'],\n    \"BsmtFinType1\": ['missing', 'Unf', 'LwQ', 'Rec', 'BLQ', 'ALQ', 'GLQ'],\n    \"BsmtFinType2\": ['missing', 'Unf', 'LwQ', 'Rec', 'BLQ', 'ALQ', 'GLQ'],\n    \"BsmtQual\": ['missing', 'Fa', 'TA', 'Gd', 'Ex'],\n    \"Electrical\": ['missing', 'Mix', 'FuseP', 'FuseF', 'FuseA', 'SBrkr'],\n    \"ExterCond\": ['missing', 'Po', 'Fa', 'TA', 'Gd', 'Ex'],\n    \"ExterQual\": ['missing', 'Fa', 'TA', 'Gd', 'Ex'],\n    \"Fence\": ['missing', 'MnWw', 'GdWo', 'MnPrv', 'GdPrv'],\n    \"FireplaceQu\": ['missing', 'Po', 'Fa', 'TA', 'Gd', 'Ex'],\n    \"Functional\": ['missing', 'Sev', 'Maj2', 'Maj1', 'Mod', 'Min2', 'Min1', 'Typ'],\n    \"GarageCond\": ['missing', 'Po', 'Fa', 'TA', 'Gd', 'Ex'],\n    \"GarageFinish\": ['missing', 'Unf', 'RFn', 'Fin'],\n    \"GarageQual\": ['missing', 'Po', 'Fa', 'TA', 'Gd', 'Ex'],\n    \"HeatingQC\": ['missing', 'Po', 'Fa', 'TA', 'Gd', 'Ex'],\n    \"KitchenQual\": ['missing', 'Fa', 'TA', 'Gd', 'Ex'],\n    \"LandContour\": ['missing', 'Low', 'Bnk', 'HLS', 'Lvl'],\n    \"LandSlope\": ['missing', 'Sev', 'Mod', 'Gtl'],\n    \"LotShape\": ['missing', 'IR3', 'IR2', 'IR1', 'Reg'],\n    \"PavedDrive\": ['missing', 'N', 'P', 'Y'],\n    \"PoolQC\": ['missing', 'Fa', 'Gd', 'Ex'],\n}\n\nords = sorted(feat_ordinal_dict.keys()) # sort alphabetically\nords_sorted = [feat_ordinal_dict[i] for i in ords]\n\nencoder_ordinal = OrdinalEncoder(\n    categories=ords_sorted,\n    dtype= np.int64,\n    handle_unknown=\"use_encoded_value\",\n    unknown_value=-1 # Considers unknown values as worse than \"missing\"\n)\n\nords_pipe = Pipeline([\n    ('imputer', SimpleImputer(strategy=\"constant\", fill_value=\"missing\")),\n    ('encoder', encoder_ordinal),\n    ('mm_scaler', MinMaxScaler())\n])","metadata":{"execution":{"iopub.status.busy":"2022-07-09T13:39:32.587622Z","iopub.execute_input":"2022-07-09T13:39:32.588004Z","iopub.status.idle":"2022-07-09T13:39:32.599755Z","shell.execute_reply.started":"2022-07-09T13:39:32.587974Z","shell.execute_reply":"2022-07-09T13:39:32.598972Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# cat pipeline\ncats = sorted(list(set(X.columns) - set(nums) - set(ords)))\n\ncats_pipe = Pipeline([\n    ('imputer', SimpleImputer(strategy='most_frequent')),\n    ('encoder', OneHotEncoder(handle_unknown='ignore'))\n])","metadata":{"execution":{"iopub.status.busy":"2022-07-09T13:39:36.261299Z","iopub.execute_input":"2022-07-09T13:39:36.261664Z","iopub.status.idle":"2022-07-09T13:39:36.267917Z","shell.execute_reply.started":"2022-07-09T13:39:36.261634Z","shell.execute_reply":"2022-07-09T13:39:36.266827Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# combine pipelines\n\nfinal_preprocessor = ColumnTransformer([\n    ('num', nums_pipe, nums),\n    ('cat', cats_pipe, cats),\n    ('ord', ords_pipe, ords)\n], remainder='drop')\n\n# test\npd.DataFrame(final_preprocessor.fit_transform(X)).head(3)","metadata":{"execution":{"iopub.status.busy":"2022-07-09T13:41:18.265296Z","iopub.execute_input":"2022-07-09T13:41:18.265669Z","iopub.status.idle":"2022-07-09T13:41:18.452131Z","shell.execute_reply.started":"2022-07-09T13:41:18.265637Z","shell.execute_reply":"2022-07-09T13:41:18.450860Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# final pipeline ready for XGBoost\nmodel_test_pipe = Pipeline([\n                    ('preproc', final_preprocessor)\n                    ])\n\npipe_xgb = make_pipeline(model_test_pipe, XGBRegressor())","metadata":{"execution":{"iopub.status.busy":"2022-07-09T13:47:50.951895Z","iopub.execute_input":"2022-07-09T13:47:50.952302Z","iopub.status.idle":"2022-07-09T13:47:50.957739Z","shell.execute_reply.started":"2022-07-09T13:47:50.952271Z","shell.execute_reply":"2022-07-09T13:47:50.956831Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 2.2 GridSearch the best parameters","metadata":{}},{"cell_type":"code","source":"kfold = StratifiedKFold(n_splits=5, shuffle=True, random_state=2)\ndef grid_search(params, random=False):\n    if random:\n        xgb_grid = RandomizedSearchCV(pipe_xgb, params, cv=kfold, n_iter=5, n_jobs=-1)\n    else:\n        xgb_grid = GridSearchCV(pipe_xgb, params, cv=kfold, n_jobs=-1)\n        \n    xgb_grid.fit(X, y)\n    xgb_best_params = xgb_grid.best_params_\n    print(f'Best params: {xgb_best_params}')\n\n    xgb_best_score = xgb_grid.best_score_\n    print(f'Training score: {xgb_best_score:.3}')","metadata":{"execution":{"iopub.status.busy":"2022-07-09T13:49:17.133390Z","iopub.execute_input":"2022-07-09T13:49:17.133782Z","iopub.status.idle":"2022-07-09T13:49:17.141051Z","shell.execute_reply.started":"2022-07-09T13:49:17.133737Z","shell.execute_reply":"2022-07-09T13:49:17.139827Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"grid = {'xgbregressor__n_estimators': [600,700,800,900,1000],\n        'xgbregressor__learning_rate': [0.01,0.05,0.1],\n        'xgbregressor__max_depth': [3,5,6],\n        'xgbregressor__gamma': [0,0.5,1,],\n        'xgbregressor__min_child_weight': [1,2,5],\n        'xgbregressor__subsample': [0.5,0.7,0.8],\n        'xgbregressor__colsample_bynode': [0.7,0.8,1],\n        'xgbregressor__colsample_bylevel': [0.7,0.8,1],\n        'xgbregressor__colsample_bytree': [0.7,0.8,1]\n       }\n\n# call the function, do a RandomizedSearch\ngrid_search(params=grid, random=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-09T13:49:19.443674Z","iopub.execute_input":"2022-07-09T13:49:19.444584Z","iopub.status.idle":"2022-07-09T13:50:56.175649Z","shell.execute_reply.started":"2022-07-09T13:49:19.444548Z","shell.execute_reply":"2022-07-09T13:50:56.173968Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 2.3 Implement the best model ","metadata":{}},{"cell_type":"code","source":"xgb = XGBRegressor(booster='gbtree',\n                    objective='reg:squarederror',\n                    max_depth=6,\n                    learning_rate=0.05,\n                    n_estimators=800,\n                    random_state=2,\n                    subsample=0.5,\n                    min_child_weight=2,\n                    gamma=0.5,\n                    colsample_bytree=1,\n                    colsample_bynode=0.7,\n                    colsample_bylevel=0.7,\n                    n_jobs=-1)","metadata":{"execution":{"iopub.status.busy":"2022-07-09T13:56:09.883180Z","iopub.execute_input":"2022-07-09T13:56:09.886421Z","iopub.status.idle":"2022-07-09T13:56:09.898165Z","shell.execute_reply.started":"2022-07-09T13:56:09.886318Z","shell.execute_reply":"2022-07-09T13:56:09.896897Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"kfold = StratifiedKFold(n_splits=5, shuffle=True, random_state=2)\npipe_xgb = make_pipeline(model_test_pipe, xgb)\nscores = cross_val_score(pipe_xgb, X, y, cv=kfold, verbose=0)\nprint(f'Accuracy: {np.round(scores, 2)}')\nprint(f'Accuracy mean: {scores.mean():.7}')","metadata":{"execution":{"iopub.status.busy":"2022-07-09T13:56:19.680195Z","iopub.execute_input":"2022-07-09T13:56:19.680659Z","iopub.status.idle":"2022-07-09T13:56:52.161176Z","shell.execute_reply.started":"2022-07-09T13:56:19.680618Z","shell.execute_reply":"2022-07-09T13:56:52.159830Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Fit the model\npipe_xgb.fit(X, y)\ny_pred_xgb = pipe_xgb.predict(X_test)","metadata":{"execution":{"iopub.status.busy":"2022-07-09T13:57:21.052471Z","iopub.execute_input":"2022-07-09T13:57:21.052898Z","iopub.status.idle":"2022-07-09T13:57:28.042835Z","shell.execute_reply.started":"2022-07-09T13:57:21.052866Z","shell.execute_reply":"2022-07-09T13:57:28.041615Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Score: 0.12584","metadata":{}}]}