{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Imports","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-13T08:44:45.824393Z","iopub.execute_input":"2022-08-13T08:44:45.824979Z","iopub.status.idle":"2022-08-13T08:44:46.603987Z","shell.execute_reply.started":"2022-08-13T08:44:45.824863Z","shell.execute_reply":"2022-08-13T08:44:46.602478Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"np.random.seed(314)","metadata":{"execution":{"iopub.status.busy":"2022-08-13T08:44:46.606887Z","iopub.execute_input":"2022-08-13T08:44:46.607397Z","iopub.status.idle":"2022-08-13T08:44:46.614279Z","shell.execute_reply.started":"2022-08-13T08:44:46.607341Z","shell.execute_reply":"2022-08-13T08:44:46.612561Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# load datasets\ntrain = pd.read_csv('/kaggle/input/house-prices-advanced-regression-techniques/train.csv')\ntest = pd.read_csv('/kaggle/input/house-prices-advanced-regression-techniques/test.csv')","metadata":{"execution":{"iopub.status.busy":"2022-08-13T08:44:46.616145Z","iopub.execute_input":"2022-08-13T08:44:46.617141Z","iopub.status.idle":"2022-08-13T08:44:46.702377Z","shell.execute_reply.started":"2022-08-13T08:44:46.617103Z","shell.execute_reply":"2022-08-13T08:44:46.700856Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(train.shape)\nprint(test.shape)","metadata":{"execution":{"iopub.status.busy":"2022-08-13T08:44:46.705232Z","iopub.execute_input":"2022-08-13T08:44:46.705691Z","iopub.status.idle":"2022-08-13T08:44:46.712729Z","shell.execute_reply.started":"2022-08-13T08:44:46.705652Z","shell.execute_reply":"2022-08-13T08:44:46.711606Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-13T08:44:46.714803Z","iopub.execute_input":"2022-08-13T08:44:46.715203Z","iopub.status.idle":"2022-08-13T08:44:46.762724Z","shell.execute_reply.started":"2022-08-13T08:44:46.715169Z","shell.execute_reply":"2022-08-13T08:44:46.761069Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-13T08:44:46.764664Z","iopub.execute_input":"2022-08-13T08:44:46.765006Z","iopub.status.idle":"2022-08-13T08:44:46.791389Z","shell.execute_reply.started":"2022-08-13T08:44:46.764977Z","shell.execute_reply":"2022-08-13T08:44:46.790183Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# EDA","metadata":{}},{"cell_type":"markdown","source":"Most of EDA already done on this notebook : https://www.kaggle.com/code/dgawlik/house-prices-eda/notebook","metadata":{}},{"cell_type":"markdown","source":"Most of the EDA can be achieved thanks to pandas_profiling. The following block is commented but it allows to watch at variables and distributions.","metadata":{}},{"cell_type":"code","source":"# from pandas_profiling import ProfileReport\n\n# profile = ProfileReport(train)\n# profile","metadata":{"execution":{"iopub.status.busy":"2022-08-13T08:44:46.792966Z","iopub.execute_input":"2022-08-13T08:44:46.793421Z","iopub.status.idle":"2022-08-13T08:44:46.802523Z","shell.execute_reply.started":"2022-08-13T08:44:46.793385Z","shell.execute_reply":"2022-08-13T08:44:46.801080Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The following block of code has purpose to pairplot all quantitative variables.\nThe resulting picture is huge (that's why it's commented) but it makes possible to detect outliers.","metadata":{}},{"cell_type":"code","source":"# sns.pairplot(train, kind=\"reg\", plot_kws={'line_kws':{'color':'red'}})","metadata":{"execution":{"iopub.status.busy":"2022-08-13T08:44:46.804432Z","iopub.execute_input":"2022-08-13T08:44:46.804974Z","iopub.status.idle":"2022-08-13T08:44:46.815095Z","shell.execute_reply.started":"2022-08-13T08:44:46.804928Z","shell.execute_reply":"2022-08-13T08:44:46.813996Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We are ploting the target and are trying various transformations.","metadata":{}},{"cell_type":"code","source":"from sklearn.preprocessing import PowerTransformer, QuantileTransformer\n\nbc = PowerTransformer(method='box-cox')\nyj = PowerTransformer(method='yeo-johnson')\nqn = QuantileTransformer(output_distribution='normal')\n\ny = train['SalePrice'].values.reshape(-1, 1)\ny_log = np.log1p(y)\ny_bc = bc.fit_transform(y)\ny_yj = yj.fit_transform(y)\ny_qn = qn.fit_transform(y)\n\nfig, axes = plt.subplots(1, 5)\nfig.set_size_inches(24, 6)\nvariables = [y, y_log, y_bc, y_yj, y_qn]\ntitles = ['SalePrice', 'log', 'box-cox', 'yeo-johnson', 'quantile-normal']\nfor i, var in enumerate(variables):\n    sns.histplot(var, ax=axes[i], legend=False)\n    axes[i].set_title(titles[i])\n\nplt.plot()","metadata":{"execution":{"iopub.status.busy":"2022-08-13T08:44:46.816702Z","iopub.execute_input":"2022-08-13T08:44:46.817078Z","iopub.status.idle":"2022-08-13T08:44:48.076188Z","shell.execute_reply.started":"2022-08-13T08:44:46.817044Z","shell.execute_reply":"2022-08-13T08:44:48.075022Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The distribution is right-skewed.\nThe transformation with log and box-cox/yeo-johnson makes it have a normal distribution shape.\nThe quantile-normal transformation also allows that, but some informations are lost due to the transformation and cross-val evaluations are less good after this transformation.","metadata":{}},{"cell_type":"code","source":"# identify outliers\nprint('LotFrontage > 300')\nprint(train[train['LotFrontage'] > 300]['Id'].to_string(index=False))\nprint('BsmtFinSF1 > 4000')\nprint(train[train['BsmtFinSF1'] > 4000]['Id'].to_string(index=False))\nprint('BsmtFinSF2 > 1400')\nprint(train[train['BsmtFinSF2'] > 1400]['Id'].to_string(index=False))\nprint('TotalBsmtSF > 6000')\nprint(train[train['TotalBsmtSF'] > 6000]['Id'].to_string(index=False))\nprint('1stFlrSF > 4000')\nprint(train[train['1stFlrSF'] > 4000]['Id'].to_string(index=False))\nprint('EnclosedPorch > 400')\nprint(train[train['EnclosedPorch'] > 400]['Id'].to_string(index=False))\nprint('MiscVal > 5000')\nprint(train[train['MiscVal'] > 5000]['Id'].to_string(index=False))\nprint('GrLivArea > 5000')\nprint(train[train['GrLivArea'] > 5000]['Id'].to_string(index=False))\n\n# Salesprices\nprint('Salesprices')\nprint(train[train['SalePrice'] > 500000][['Id', 'SalePrice']]\n    .sort_values(by='SalePrice', ascending=False).to_string(index=False))","metadata":{"execution":{"iopub.status.busy":"2022-08-13T08:44:48.081465Z","iopub.execute_input":"2022-08-13T08:44:48.082860Z","iopub.status.idle":"2022-08-13T08:44:48.109244Z","shell.execute_reply.started":"2022-08-13T08:44:48.082806Z","shell.execute_reply":"2022-08-13T08:44:48.107979Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We have detected some outliers, we get the Ids of these outliers and will process them.","metadata":{}},{"cell_type":"markdown","source":"# Pre-processing","metadata":{}},{"cell_type":"markdown","source":"## Outliers","metadata":{}},{"cell_type":"code","source":"def remove_outliers(train):\n    # get Ids after EDA\n    # execute eda_outliers.py for outliers analysis\n\n    ids = [\n        692, 1183,  # SalePrice > 700000\n        935,        # LotFrontage > 300\n        1299,       # BsmtFinSF1 > 4000, TotalBsmtSF > 6000,\n                    # 1stFlrSF > 4000, GrLivArea > 5000\n        323,        # BsmtFinSF2 > 1400\n        198,        # EnclosedPorch > 400\n        347, 1231   # MiscVal > 5000\n    ]\n\n    return train[~train['Id'].isin(ids)]","metadata":{"execution":{"iopub.status.busy":"2022-08-13T08:44:48.111270Z","iopub.execute_input":"2022-08-13T08:44:48.112379Z","iopub.status.idle":"2022-08-13T08:44:48.120285Z","shell.execute_reply.started":"2022-08-13T08:44:48.112331Z","shell.execute_reply":"2022-08-13T08:44:48.118858Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(train.shape)\ntrain = remove_outliers(train)\nprint(train.shape)","metadata":{"execution":{"iopub.status.busy":"2022-08-13T08:44:48.122216Z","iopub.execute_input":"2022-08-13T08:44:48.124363Z","iopub.status.idle":"2022-08-13T08:44:48.140411Z","shell.execute_reply.started":"2022-08-13T08:44:48.124299Z","shell.execute_reply":"2022-08-13T08:44:48.139063Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Feature engineering","metadata":{}},{"cell_type":"markdown","source":"We are reproducing feature engineering from this notebook : https://www.kaggle.com/code/dgawlik/house-prices-eda/notebook","metadata":{}},{"cell_type":"code","source":"from sklearn.base import TransformerMixin, BaseEstimator\n\n\nclass DataframeFunctionTransformer(TransformerMixin, BaseEstimator):\n    \"\"\"\n    Do transformations regarding a specific function\n    \"\"\"\n    def __init__(self, func):\n        self.func = func\n\n    def transform(self, input_df, **transform_params):\n        \"\"\"\n        performs the function transformation\n        \"\"\"\n        return self.func(input_df)\n\n    def fit(self, X, y=None, **fit_params):\n        \"\"\"\n        fit method\n        \"\"\"\n        return self\n\nclass DenseTransformer(TransformerMixin, BaseEstimator):\n    \"\"\"\n    Transforms parse matrix into numpy array\n    \"\"\"\n    def fit(self, X, y=None, **fit_params):\n        \"\"\"\n        fit method\n        \"\"\"\n        return self\n\n    def transform(self, X, y=None, **fit_params):\n        \"\"\"\n        transform method\n        \"\"\"\n        return X.toarray()\n\ndef fillna(df):\n    \"\"\"fill na values on dataframe\n\n    Args:\n        df (pandas.DataFrame): original dataframe\n    Returns:\n        pandas.DataFrame: copy of original dataframe with filled values\n    \"\"\"\n    copy = df.copy()\n    values = {\n        'Electrical': 'SBrkr',\n        'MasVnrType': 'None',\n        'MasVnrArea': 0,\n    }\n\n    for k, v in values.items():\n        copy[k].fillna(v, inplace=True)\n\n    copy['GarageYrBlt'].fillna(copy['YearBuilt'], inplace=True)\n\n    for k in ['BsmtQual', 'BsmtCond', 'BsmtFinType1', 'BsmtExposure',\n        'BsmtFinType2', 'GarageType', 'GarageFinish', 'GarageQual',\n        'GarageCond', 'FireplaceQu', 'Fence', 'Alley', 'MiscFeature',\n        'PoolQC']:\n        copy[k].fillna('NA', inplace=True)\n\n    return copy\n\ndef bool_features(df):\n    \"\"\"Create bool features for some variables\n\n    Args:\n        df (pandas.DataFrame): dataframe\n\n    Returns:\n        pandas.DataFrame: copy of original dataframe with added bool features\n    \"\"\"\n    copy = df.copy()\n\n    copy['HasBasement'] = copy['TotalBsmtSF'].apply(lambda x: 1 if x > 0 else 0)\n    copy['HasGarage'] = copy['GarageArea'].apply(lambda x: 1 if x > 0 else 0)\n    copy['Has2ndFloor'] = copy['2ndFlrSF'].apply(lambda x: 1 if x > 0 else 0)\n    copy['HasMasVnr'] = copy['MasVnrArea'].apply(lambda x: 1 if x > 0 else 0)\n    copy['HasWoodDeck'] = copy['WoodDeckSF'].apply(lambda x: 1 if x > 0 else 0)\n    copy['HasPorch'] = copy['OpenPorchSF'].apply(lambda x: 1 if x > 0 else 0)\n    copy['HasPool'] = copy['PoolArea'].apply(lambda x: 1 if x > 0 else 0)\n    copy['IsNew'] = copy['YearBuilt'].apply(lambda x: 1 if x > 2000 else 0)\n\n    return copy\n","metadata":{"execution":{"iopub.status.busy":"2022-08-13T08:44:48.142240Z","iopub.execute_input":"2022-08-13T08:44:48.143541Z","iopub.status.idle":"2022-08-13T08:44:48.163146Z","shell.execute_reply.started":"2022-08-13T08:44:48.143461Z","shell.execute_reply":"2022-08-13T08:44:48.161288Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import \\\n    OrdinalEncoder, PolynomialFeatures, OneHotEncoder, FunctionTransformer, \\\n    StandardScaler\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.compose import ColumnTransformer\nfrom sklearn.impute import KNNImputer\n\nnum_cols = [col for col in train.columns if train.dtypes[col] != 'object']\nnum_cols.remove('SalePrice')\nnum_cols.remove('Id')\nnum_cols.remove('MSSubClass') # MSSubClass is actually categorical\ncat_cols = [col for col in train.columns if train.dtypes[col] == 'object']\ncat_cols.append('MSSubClass')\n\nord_scores = ['NA', 'Po', 'Fa', 'TA', 'Gd', 'Ex']\nscores_cat_cols = ['ExterQual', 'ExterCond', 'BsmtQual', 'BsmtCond',\n    'HeatingQC', 'KitchenQual', 'FireplaceQu', 'GarageQual',\n    'GarageCond', 'PoolQC']\n\nother_cat_cols = list(\n    set(cat_cols)\n    - set(scores_cat_cols)\n    - set(['Electrical', 'CentralAir', 'GarageFinish', 'PavedDrive']))\n\nlog_num_cols = [\n    'GrLivArea', '1stFlrSF', '2ndFlrSF', 'TotalBsmtSF',\n    'LotArea', 'LotFrontage', 'KitchenAbvGr', 'GarageArea'\n]\n\nquad_num_cols = [\n    'OverallQual', 'YearBuilt', 'YearRemodAdd',\n    '2ndFlrSF', 'GrLivArea',\n]\n\ncolumns = ColumnTransformer(transformers = [\n    ('scores_transform', OrdinalEncoder(\n        categories=len(scores_cat_cols) * [ord_scores],\n        handle_unknown='use_encoded_value', unknown_value=-1), scores_cat_cols),\n    ('log_transform', FunctionTransformer(np.log1p), log_num_cols),\n    ('quad_transform', PolynomialFeatures(degree=2), quad_num_cols),\n    ('electrical', OrdinalEncoder(\n        categories=[['Mix', 'FuseP', 'FuseF', 'FuseA', 'SBrkr']]), ['Electrical']),\n    ('central_air', OrdinalEncoder(\n        categories=[['N', 'Y']]), ['CentralAir']),\n    ('garage_finish', OrdinalEncoder(\n        categories=[['NA', 'Unf', 'RFn', 'Fin']]), ['GarageFinish']),\n    ('paved_drive', OrdinalEncoder(\n        categories=[['N', 'P', 'Y']],\n        handle_unknown='use_encoded_value', unknown_value=-1), ['PavedDrive']),\n    ('one_hot', OneHotEncoder(handle_unknown='ignore'), other_cat_cols),\n])\n\npreprocessor = Pipeline([\n    ('fillna', DataframeFunctionTransformer(fillna)),\n    ('bool_features', DataframeFunctionTransformer(bool_features)),\n    ('columns', columns),\n    ('to_dense', DenseTransformer()),\n    ('standard', StandardScaler()),\n    ('impute', KNNImputer()),\n])","metadata":{"execution":{"iopub.status.busy":"2022-08-13T08:44:48.165818Z","iopub.execute_input":"2022-08-13T08:44:48.167550Z","iopub.status.idle":"2022-08-13T08:44:48.426341Z","shell.execute_reply.started":"2022-08-13T08:44:48.167463Z","shell.execute_reply":"2022-08-13T08:44:48.425034Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train = preprocessor.fit_transform(train.drop(['Id', 'SalePrice'], axis=1))\nprint(X_train.shape)\ny_train = train['SalePrice']","metadata":{"execution":{"iopub.status.busy":"2022-08-13T08:44:48.427742Z","iopub.execute_input":"2022-08-13T08:44:48.428087Z","iopub.status.idle":"2022-08-13T08:44:48.771194Z","shell.execute_reply.started":"2022-08-13T08:44:48.428057Z","shell.execute_reply":"2022-08-13T08:44:48.769756Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Models testing","metadata":{}},{"cell_type":"code","source":"import time\nfrom optuna.integration import OptunaSearchCV\n\ndef search_cv(X, y, model, param_distributions={}, random_state=None,\n    n_splits=5, n_jobs=-1, n_trials=10, scoring='neg_mean_squared_log_error'):\n    \n    search_args = {\n        'estimator': model,\n        'param_distributions': param_distributions,\n        'cv': n_splits,\n        'scoring': scoring,\n        'n_jobs': n_jobs,\n        'n_trials': n_trials,\n        'random_state': random_state,\n    }\n\n    opt_search_cv = OptunaSearchCV(**search_args)\n\n    start = time.time()\n    opt_search_cv.fit(X, y.values.ravel())\n    end = time.time()\n\n    print('time:', end - start, 'seconds')\n\n    return opt_search_cv, end - start\n","metadata":{"execution":{"iopub.status.busy":"2022-08-13T08:44:48.773012Z","iopub.execute_input":"2022-08-13T08:44:48.773377Z","iopub.status.idle":"2022-08-13T08:44:49.582543Z","shell.execute_reply.started":"2022-08-13T08:44:48.773345Z","shell.execute_reply":"2022-08-13T08:44:49.581412Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.compose import TransformedTargetRegressor\nfrom optuna.distributions import \\\n    IntUniformDistribution, UniformDistribution, LogUniformDistribution, CategoricalDistribution\n","metadata":{"execution":{"iopub.status.busy":"2022-08-13T08:44:49.584172Z","iopub.execute_input":"2022-08-13T08:44:49.587017Z","iopub.status.idle":"2022-08-13T08:44:49.592655Z","shell.execute_reply.started":"2022-08-13T08:44:49.586974Z","shell.execute_reply":"2022-08-13T08:44:49.591413Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tr = PowerTransformer(method='box-cox')\ntr.fit(train['SalePrice'].values.reshape(-1, 1))","metadata":{"execution":{"iopub.status.busy":"2022-08-13T08:44:49.594277Z","iopub.execute_input":"2022-08-13T08:44:49.595459Z","iopub.status.idle":"2022-08-13T08:44:49.612210Z","shell.execute_reply.started":"2022-08-13T08:44:49.595417Z","shell.execute_reply":"2022-08-13T08:44:49.610818Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def create_submission_file(model, name):\n    X_test = preprocessor.transform(test.drop('Id', axis=1))\n    print(X_test.shape)\n    y_pred = model.predict(X_test)\n\n    submission = pd.DataFrame({'Id': test.Id, 'SalePrice': y_pred})\n\n    # dirty trick : replace inf value with max value\n    m = submission.loc[submission['SalePrice'] != np.inf, 'SalePrice'].max()\n    submission['SalePrice'].replace(np.inf,m,inplace=True)\n\n    submission.to_csv(f'{name}.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-08-13T08:44:49.614125Z","iopub.execute_input":"2022-08-13T08:44:49.614593Z","iopub.status.idle":"2022-08-13T08:44:49.622970Z","shell.execute_reply.started":"2022-08-13T08:44:49.614555Z","shell.execute_reply":"2022-08-13T08:44:49.621602Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Catboost regressor","metadata":{}},{"cell_type":"code","source":"from catboost import CatBoostRegressor\n\nparams = {\n    'regressor__learning_rate': LogUniformDistribution(1e-3, 1e-1),\n    'regressor__l2_leaf_reg': LogUniformDistribution(1e-4, 1e-1),\n    'regressor__max_depth': IntUniformDistribution(1, 8),\n    'regressor__min_data_in_leaf': IntUniformDistribution(1, 300),\n}\n\nmodel = TransformedTargetRegressor(\n    regressor=CatBoostRegressor(\n        silent=True, thread_count=1, n_estimators=1000),\n    func=tr.transform,\n    inverse_func=tr.inverse_transform)\n\nscv_bc_catboost, train_time = search_cv(X_train, y_train, model, params, n_trials=32)\n\nprint('score:', scv_bc_catboost.best_score_, 'params:', scv_bc_catboost.best_params_, f'took {train_time} seconds')\n\ncreate_submission_file(scv_bc_catboost, 'catboost')","metadata":{"execution":{"iopub.status.busy":"2022-08-13T08:44:49.624867Z","iopub.execute_input":"2022-08-13T08:44:49.626204Z","iopub.status.idle":"2022-08-13T08:55:40.030808Z","shell.execute_reply.started":"2022-08-13T08:44:49.626143Z","shell.execute_reply":"2022-08-13T08:55:40.029014Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Multi-Layer Perceptron","metadata":{}},{"cell_type":"code","source":"from sklearn.neural_network import MLPRegressor\n\nparams = {\n    'regressor__hidden_layer_sizes': CategoricalDistribution([\n        (3,), (4,), (5,), (10,),\n        (3,3), (4,4), (5,5), (10,10),\n        (3,3,3), (4,4,4), (5,5,5), (10,10,3),\n        ]),\n    'regressor__alpha': LogUniformDistribution(1e-5, 1e-1),\n}\n\nmodel = TransformedTargetRegressor(\n    regressor=MLPRegressor(max_iter=1000, solver='sgd', learning_rate_init=1e-2),\n    func=tr.transform, inverse_func=tr.inverse_transform)\n\nscv_bc_mlp, train_time = search_cv(X_train, y_train, model, params, n_trials=64)\n\nprint('score:', scv_bc_mlp.best_score_, 'params:', scv_bc_mlp.best_params_, f'took {train_time} seconds')\n\ncreate_submission_file(scv_bc_mlp, 'mlp')","metadata":{"execution":{"iopub.status.busy":"2022-08-13T08:55:40.032651Z","iopub.execute_input":"2022-08-13T08:55:40.033189Z","iopub.status.idle":"2022-08-13T09:06:13.512513Z","shell.execute_reply.started":"2022-08-13T08:55:40.033150Z","shell.execute_reply":"2022-08-13T09:06:13.510897Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## SVM Regressor ","metadata":{}},{"cell_type":"code","source":"from sklearn.svm import SVR\n\nparams = {\n    'regressor__C': LogUniformDistribution(1e-3, 1e3),\n    'regressor__gamma': LogUniformDistribution(1e-6, 1e3),\n    'regressor__epsilon': LogUniformDistribution(1e-3, 1e0),\n}\n\nmodel = TransformedTargetRegressor(\n    regressor=SVR(), func=np.log1p, inverse_func=np.expm1)\n\nscv_log_svm, train_time = search_cv(X_train, y_train, model, params, n_trials=128)\n\nprint('score:', scv_log_svm.best_score_, 'params:', scv_log_svm.best_params_, f'took {train_time} seconds')\n\ncreate_submission_file(scv_log_svm, 'svm')","metadata":{"execution":{"iopub.status.busy":"2022-08-13T09:06:13.520043Z","iopub.execute_input":"2022-08-13T09:06:13.520982Z","iopub.status.idle":"2022-08-13T09:07:46.391199Z","shell.execute_reply.started":"2022-08-13T09:06:13.520919Z","shell.execute_reply":"2022-08-13T09:07:46.389888Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Lasso","metadata":{}},{"cell_type":"code","source":"from sklearn.linear_model import Lasso\n\nparams = {\n    'regressor__alpha': LogUniformDistribution(1e-6, 1e3),\n}\n\nmodel = TransformedTargetRegressor(\n    regressor=Lasso(tol=1e-1, max_iter=1e4), func=np.log1p, inverse_func=np.expm1)\n\nscv_log_lasso, train_time = search_cv(X_train, y_train, model, params, n_trials=128)\n\nprint('score:', scv_log_lasso.best_score_, 'params:', scv_log_lasso.best_params_, f'took {train_time} seconds')\n\ncreate_submission_file(scv_log_lasso, 'lasso')","metadata":{"execution":{"iopub.status.busy":"2022-08-13T09:07:46.393334Z","iopub.execute_input":"2022-08-13T09:07:46.393873Z","iopub.status.idle":"2022-08-13T09:07:59.456809Z","shell.execute_reply.started":"2022-08-13T09:07:46.393838Z","shell.execute_reply":"2022-08-13T09:07:59.451791Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Stacking models","metadata":{}},{"cell_type":"code","source":"def clean_keys(d):\n    return {k.replace('regressor__', ''):v for k,v in d.items()}","metadata":{"execution":{"iopub.status.busy":"2022-08-13T09:07:59.458818Z","iopub.execute_input":"2022-08-13T09:07:59.462569Z","iopub.status.idle":"2022-08-13T09:07:59.471527Z","shell.execute_reply.started":"2022-08-13T09:07:59.462471Z","shell.execute_reply":"2022-08-13T09:07:59.470036Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.ensemble import StackingRegressor\nfrom sklearn.linear_model import ElasticNetCV\n\nbc_catboost = TransformedTargetRegressor(\n    regressor=CatBoostRegressor(\n        silent=True, thread_count=1, n_estimators=1000,\n        **clean_keys(scv_bc_catboost.best_params_)),\n    func=tr.transform,\n    inverse_func=tr.inverse_transform)\n\nbc_mlp = TransformedTargetRegressor(\n    regressor=MLPRegressor(\n        max_iter=1000, solver='sgd',\n        learning_rate_init=1e-2,\n        **clean_keys(scv_bc_mlp.best_params_)),\n    func=tr.transform, inverse_func=tr.inverse_transform)\n\nlog_svm = TransformedTargetRegressor(\n    regressor=SVR(**clean_keys(scv_log_svm.best_params_)),\n    func=np.log1p, inverse_func=np.expm1)\n\nlog_lasso = TransformedTargetRegressor(\n    regressor=Lasso(tol=1e-1, max_iter=1e4,\n                    **clean_keys(scv_log_lasso.best_params_)),\n    func=np.log1p, inverse_func=np.expm1)\n\nestimators = [\n    ('bc_catboost', bc_catboost),\n    ('bc_mlp', bc_mlp),\n    ('log_svm', log_svm),\n    ('log_lasso', log_lasso),\n]\n\nmodel = StackingRegressor(\n    estimators=estimators, final_estimator=ElasticNetCV(cv=5))\n\nparams = {\n    'final_estimator__l1_ratio': UniformDistribution(0.0, 1.0),\n    'final_estimator__eps': LogUniformDistribution(1e-6, 1e-3),\n}\n\nscv_stack, train_time = search_cv(X_train, y_train, model, params, n_trials=8)\n\nprint('score:', scv_stack.best_score_, 'params:', scv_stack.best_params_, f'took {train_time} seconds')\n\ncreate_submission_file(scv_stack, 'stack')","metadata":{"execution":{"iopub.status.busy":"2022-08-13T09:07:59.473366Z","iopub.execute_input":"2022-08-13T09:07:59.474756Z","iopub.status.idle":"2022-08-13T09:22:07.796411Z","shell.execute_reply.started":"2022-08-13T09:07:59.474703Z","shell.execute_reply":"2022-08-13T09:22:07.794631Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Comparison of cross-val scores","metadata":{}},{"cell_type":"code","source":"print('Catboost:', scv_bc_catboost.best_score_)\nprint('MLP:', scv_bc_mlp.best_score_)\nprint('SVM:', scv_log_svm.best_score_)\nprint('Lasso:', scv_log_lasso.best_score_)\nprint('Stack:', scv_stack.best_score_)","metadata":{"execution":{"iopub.status.busy":"2022-08-13T09:22:07.805344Z","iopub.execute_input":"2022-08-13T09:22:07.809888Z","iopub.status.idle":"2022-08-13T09:22:07.831711Z","shell.execute_reply.started":"2022-08-13T09:22:07.809793Z","shell.execute_reply":"2022-08-13T09:22:07.829409Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Stack improved the best model on cross-val score","metadata":{}}]}