{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport seaborn as sns\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.decomposition import PCA\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.preprocessing import MinMaxScaler, OneHotEncoder\nfrom sklearn.compose import ColumnTransformer\nfrom sklearn.ensemble import ExtraTreesRegressor\nimport numpy as np\nfrom sklearn.ensemble import GradientBoostingRegressor\nfrom sklearn.metrics import mean_squared_error\nfrom math import sqrt\nimport catboost\nfrom sklearn.ensemble import RandomForestRegressor\nfrom sklearn.feature_selection import mutual_info_regression ","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-02-21T16:00:13.606105Z","iopub.execute_input":"2022-02-21T16:00:13.607166Z","iopub.status.idle":"2022-02-21T16:00:15.123748Z","shell.execute_reply.started":"2022-02-21T16:00:13.607035Z","shell.execute_reply":"2022-02-21T16:00:15.123019Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv('../input/house-prices-advanced-regression-techniques/train.csv')\ntest = pd.read_csv('../input/house-prices-advanced-regression-techniques/test.csv')\nsample_submission = pd.read_csv('../input/house-prices-advanced-regression-techniques/sample_submission.csv')","metadata":{"execution":{"iopub.status.busy":"2022-02-21T16:12:46.912488Z","iopub.execute_input":"2022-02-21T16:12:46.913241Z","iopub.status.idle":"2022-02-21T16:12:46.967074Z","shell.execute_reply.started":"2022-02-21T16:12:46.913195Z","shell.execute_reply":"2022-02-21T16:12:46.966346Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head()","metadata":{"execution":{"iopub.status.busy":"2022-02-21T16:12:49.892014Z","iopub.execute_input":"2022-02-21T16:12:49.892707Z","iopub.status.idle":"2022-02-21T16:12:49.929608Z","shell.execute_reply.started":"2022-02-21T16:12:49.892661Z","shell.execute_reply":"2022-02-21T16:12:49.928715Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_submission.head()","metadata":{"execution":{"iopub.status.busy":"2022-02-21T16:12:54.479462Z","iopub.execute_input":"2022-02-21T16:12:54.479749Z","iopub.status.idle":"2022-02-21T16:12:54.488612Z","shell.execute_reply.started":"2022-02-21T16:12:54.479715Z","shell.execute_reply":"2022-02-21T16:12:54.488027Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Feature Engineering","metadata":{}},{"cell_type":"code","source":"# Let's see data types and number of missing values for each column\ntrain.info()","metadata":{"execution":{"iopub.status.busy":"2022-02-21T16:13:01.117472Z","iopub.execute_input":"2022-02-21T16:13:01.117941Z","iopub.status.idle":"2022-02-21T16:13:01.157521Z","shell.execute_reply.started":"2022-02-21T16:13:01.117901Z","shell.execute_reply":"2022-02-21T16:13:01.15676Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#  drop columns with many missing values, target variable and Id column\nX = train.drop(labels=['Alley', 'FireplaceQu', 'PoolQC', 'Fence', 'MiscFeature', 'SalePrice', 'Id'], axis=1)\n# stay only target - SalePrice\ny = train['SalePrice']","metadata":{"execution":{"iopub.status.busy":"2022-02-21T16:13:04.612325Z","iopub.execute_input":"2022-02-21T16:13:04.612841Z","iopub.status.idle":"2022-02-21T16:13:04.619922Z","shell.execute_reply.started":"2022-02-21T16:13:04.612806Z","shell.execute_reply":"2022-02-21T16:13:04.618962Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# fill mising values by -1\nX.fillna(-1, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-02-21T16:13:07.201893Z","iopub.execute_input":"2022-02-21T16:13:07.202544Z","iopub.status.idle":"2022-02-21T16:13:07.219369Z","shell.execute_reply.started":"2022-02-21T16:13:07.2025Z","shell.execute_reply":"2022-02-21T16:13:07.218517Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# let's drop columns where mutual information is zero\ndef drop_cols_with_zero_mi(X, y) -> pd.DataFrame:\n    X = X.copy()\n    categorical_features = [x for x in X.columns if X[x].dtype==\"object\"]\n    for colname in X[categorical_features]:\n        X[colname], _ = X[colname].factorize()\n    \n    discrete_features = [pd.api.types.is_integer_dtype(t) for t in X.dtypes]\n    mi_scores = mutual_info_regression(X, y, discrete_features=discrete_features)\n    mi_scores = pd.Series(mi_scores, index=X.columns)\n    return X.loc[:, mi_scores > 0.0]","metadata":{"execution":{"iopub.status.busy":"2022-02-21T16:13:10.555006Z","iopub.execute_input":"2022-02-21T16:13:10.555496Z","iopub.status.idle":"2022-02-21T16:13:10.562573Z","shell.execute_reply.started":"2022-02-21T16:13:10.555463Z","shell.execute_reply":"2022-02-21T16:13:10.561742Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = drop_cols_with_zero_mi(X, y)","metadata":{"execution":{"iopub.status.busy":"2022-02-21T16:13:33.534961Z","iopub.execute_input":"2022-02-21T16:13:33.535276Z","iopub.status.idle":"2022-02-21T16:13:36.330606Z","shell.execute_reply.started":"2022-02-21T16:13:33.535243Z","shell.execute_reply":"2022-02-21T16:13:36.329589Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# split into train and test \nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.3, random_state=42)","metadata":{"execution":{"iopub.status.busy":"2022-02-21T16:13:39.144553Z","iopub.execute_input":"2022-02-21T16:13:39.144842Z","iopub.status.idle":"2022-02-21T16:13:39.152534Z","shell.execute_reply.started":"2022-02-21T16:13:39.144806Z","shell.execute_reply":"2022-02-21T16:13:39.151754Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# build pipeline\n# normalization and pca for numeric features\nnumeric_features = X_train._get_numeric_data().columns\nnumeric_transformer = Pipeline(steps=[('scaler', MinMaxScaler()),\n                                      ('pca', PCA(svd_solver='arpack'))])\n\n# coding for categorial features\ncategorical_features = [x for x in X_train.columns if X_train[x].dtype==\"object\"]\nX_train[categorical_features] = X_train[categorical_features].astype(str)\ncategorical_transformer = OneHotEncoder(handle_unknown='ignore')\n\n# combining categorical and numerical features into one pipeline\npreprocessor = ColumnTransformer(\n    transformers=[\n        ('num', numeric_transformer, numeric_features),\n        ('cat', categorical_transformer, categorical_features)])\n\n# processed data pipeline and models ExtraTreesRegressor\npipe = Pipeline([\n                ('preprocessor', preprocessor),\n                ('regression', ExtraTreesRegressor(n_estimators=2, min_samples_leaf=3, max_depth=5, random_state=120))\n            ])","metadata":{"execution":{"iopub.status.busy":"2022-02-21T16:13:41.165098Z","iopub.execute_input":"2022-02-21T16:13:41.165381Z","iopub.status.idle":"2022-02-21T16:13:41.18026Z","shell.execute_reply.started":"2022-02-21T16:13:41.165353Z","shell.execute_reply":"2022-02-21T16:13:41.17944Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pipe.fit(X_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2022-02-21T16:13:44.296886Z","iopub.execute_input":"2022-02-21T16:13:44.297227Z","iopub.status.idle":"2022-02-21T16:13:44.478003Z","shell.execute_reply.started":"2022-02-21T16:13:44.297197Z","shell.execute_reply":"2022-02-21T16:13:44.477138Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions_extra_tree = pipe.predict(X_test)","metadata":{"execution":{"iopub.status.busy":"2022-02-21T16:13:47.913109Z","iopub.execute_input":"2022-02-21T16:13:47.913519Z","iopub.status.idle":"2022-02-21T16:13:47.924116Z","shell.execute_reply.started":"2022-02-21T16:13:47.913491Z","shell.execute_reply":"2022-02-21T16:13:47.923074Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# calculate RMSE for predictions\nrmse = sqrt(mean_squared_error(y_test, predictions_extra_tree))\nprint('RMSE with ExtraTreesRegressor:', rmse)","metadata":{"execution":{"iopub.status.busy":"2022-02-21T16:13:49.988111Z","iopub.execute_input":"2022-02-21T16:13:49.988364Z","iopub.status.idle":"2022-02-21T16:13:49.995316Z","shell.execute_reply.started":"2022-02-21T16:13:49.988339Z","shell.execute_reply":"2022-02-21T16:13:49.994304Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# build catboost model\npipe_catboost = Pipeline([\n                ('preprocessor', preprocessor),\n                ('regression', catboost.CatBoostRegressor(loss_function='RMSE'))\n            ])\npipe_catboost.fit(X_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2022-02-21T16:13:51.996032Z","iopub.execute_input":"2022-02-21T16:13:51.996306Z","iopub.status.idle":"2022-02-21T16:14:01.72089Z","shell.execute_reply.started":"2022-02-21T16:13:51.996275Z","shell.execute_reply":"2022-02-21T16:14:01.720017Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"catboost_predictions = pipe_catboost.predict(X_test)\ncatboost_rmse = sqrt(mean_squared_error(y_test, catboost_predictions))\nprint('RMSE with CatBoost:', catboost_rmse)","metadata":{"execution":{"iopub.status.busy":"2022-02-21T16:14:06.121693Z","iopub.execute_input":"2022-02-21T16:14:06.122341Z","iopub.status.idle":"2022-02-21T16:14:06.18132Z","shell.execute_reply.started":"2022-02-21T16:14:06.122301Z","shell.execute_reply":"2022-02-21T16:14:06.180294Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# build random forest model\npipe_rf = Pipeline([\n                ('preprocessor', preprocessor),\n                ('regression', RandomForestRegressor())\n            ])\npipe_rf.fit(X_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2022-02-21T16:14:09.20506Z","iopub.execute_input":"2022-02-21T16:14:09.205348Z","iopub.status.idle":"2022-02-21T16:14:13.465033Z","shell.execute_reply.started":"2022-02-21T16:14:09.205317Z","shell.execute_reply":"2022-02-21T16:14:13.464172Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rf_predictions = pipe_rf.predict(X_test)\nrf_rmse = sqrt(mean_squared_error(y_test, rf_predictions))\nprint('RMSE with RandomForestRegressor:', rf_rmse)","metadata":{"execution":{"iopub.status.busy":"2022-02-21T16:14:16.569589Z","iopub.execute_input":"2022-02-21T16:14:16.570197Z","iopub.status.idle":"2022-02-21T16:14:16.610859Z","shell.execute_reply.started":"2022-02-21T16:14:16.570145Z","shell.execute_reply":"2022-02-21T16:14:16.61002Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# The best result is achieved with catboost model\n# test = test.fillna(0)\n# catboost_test_predict = pipe_catboost.predict(test)\n# sample_submission = pd.DataFrame(columns=['Id', 'SalePrice'])\n# sample_submission['Id'] = test['Id']\n# sample_submission['SalePrice'] = catboost_test_predict\n# sample_submission.to_csv('submission.csv', index=False)","metadata":{},"execution_count":null,"outputs":[]}]}