{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport warnings\nimport xgboost as xgb\nfrom sklearn.preprocessing import OrdinalEncoder, OneHotEncoder, \\\n    QuantileTransformer, PolynomialFeatures, MinMaxScaler\nfrom sklearn.compose import ColumnTransformer\nfrom sklearn.tree import DecisionTreeRegressor\nfrom sklearn.metrics import make_scorer\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.linear_model import Ridge\nfrom sklearn.model_selection import cross_validate, GridSearchCV\nfrom sklearn.svm import SVR\nsns.set()\nwarnings.filterwarnings('ignore')\n%matplotlib inline","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e4ac8d440139c9610e28009994e0b8a7b47f6592","_kg_hide-input":false,"_kg_hide-output":false},"cell_type":"code","source":"# Helper functions\n\ndef countna(df):\n    \"\"\"Return the count of null values per column\"\"\"\n    s = df.isna().sum()\n    return s[s > 0]\n\ndef plot_pointplot(df, y_vars, x_vars, height, aspect):\n    \"\"\"Plot a series of point plots and rotate the x labels\"\"\"\n    g = sns.PairGrid(df, y_vars=y_vars, x_vars=x_vars, height=height,\n                     aspect=aspect)\n    g.map(sns.pointplot)\n    for ax in g.axes.ravel():\n        ax.set_xticklabels(ax.get_xticklabels(), rotation=90)\n    plt.show()\n\ndef nested_cross_validation(estimator, param_grid, X_train, y_train, scoring):\n    \"\"\"Cross validate the specified estimator using 5x2 nested cross \n    validation. Print the test scores on the screen and return the array\n    with all the scoring data.\"\"\"\n    gs = GridSearchCV(estimator, param_grid, cv=2)\n    scores = cross_validate(gs, X_train, y_train, cv=5, scoring=scoring)\n    for key in scoring.keys():\n        s = scores['test_'+key]\n        print('CV %s accuracy: %.4f +/- %.4f' % (key, np.mean(s), np.std(s)))\n    return scores\n\ndef rmse(y, y_pred):\n    \"\"\"Calculate RMSE (Root Mean Square Error)\"\"\"\n    return np.sqrt(np.mean((y_pred - y)**2))\n\n# Make a custom scorer for RMSE (not available in sklearn)\nrmse_scorer = make_scorer(rmse, greater_is_better=False)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"1ae34cfe8bf1332447d8ef130dcd63ed93f181c5"},"cell_type":"markdown","source":"# Initial Data Exploration\n\nFirst let's identify the types of features and the number of missing values in the training and test datasets"},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"# Read the datasets\ntrain = pd.read_csv('../input/train.csv')\ntest = pd.read_csv('../input/test.csv')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"57c4f4c9959e1a9655450f4fd1a0ff80af71640d"},"cell_type":"code","source":"# Count of features by data type\ntrain.dtypes.value_counts()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f02927d3229f024dcaf137821589f4ec0b924e04"},"cell_type":"code","source":"# Features of type object (nominal)\ntrain.columns[train.dtypes == np.object]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"3b11ea27f13e6902762a94799b939682fed663f9"},"cell_type":"code","source":"# Features of type int64 (should contain only ordinal)\ntrain.columns[train.dtypes == np.int64]","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"a17f6abe1c5008117ea546ba8540176d9099655f"},"cell_type":"markdown","source":"There is a mixture of ordinal and numerical (continous & discrete) features using the int64 data type. It would be easier to distinguish them if their types were different:\n* Ordinal = int64\n* Numerical = float64\n\nLet's fix that later on in the preprocessing step\n"},{"metadata":{"trusted":true,"_uuid":"eb3be83c355893489916995c5f63c6b745b3d292"},"cell_type":"code","source":"# Features of type float64 (numerical : continous & discrete)\ntrain.columns[train.dtypes == np.float64]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"0c0dec1077830e87ed18591b1e2cb6b50047dadf"},"cell_type":"code","source":"# Check for missing numeric values in the train dataset\nprint('Missing values in train dataset:') \ncountna(train)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"83780402bd2fe298ffcd3065dc5b46d2c510390f"},"cell_type":"code","source":"# Check for missing numeric values in the test dataset\nprint('Missing values in test dataset:')\ncountna(test)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"9cc710fd866efbb29ce5c9843beb2ff618451289"},"cell_type":"markdown","source":"There are many missing values in the training and test datasets across the different types of features (nominal, ordinal & numeric). We will need to fill the mising values for each feature type in the preprocessing step. "},{"metadata":{"_uuid":"c8fff9abc1f198a752d6646ae58d10d859bdb9f4"},"cell_type":"markdown","source":"# Data Preprocessing\n\nLet's fix the data types of some features and fill the missing values."},{"metadata":{"trusted":true,"_uuid":"5847c65a83aea9e730c2e10974db37fabc614603"},"cell_type":"code","source":"def preprocess_features(housing_data):\n    # Make a copy of the dataframe\n    preprocessed = housing_data.copy()\n    \n    # Data type conversions\n    \n    # The following features are numerical (continous or discrete) but have\n    # the type int64, let's convert them to float64\n    numerical_feat = [\n        'LotArea', 'YearBuilt', 'YearRemodAdd', 'BsmtFinSF1', \n        'BsmtFinSF2', 'BsmtUnfSF', 'TotalBsmtSF', '1stFlrSF',\n        '2ndFlrSF', 'LowQualFinSF', 'GrLivArea', 'BsmtFullBath',\n        'BsmtHalfBath', 'FullBath', 'HalfBath', 'BedroomAbvGr',\n        'KitchenAbvGr', 'TotRmsAbvGrd', 'Fireplaces', 'GarageCars',\n        'GarageArea', 'WoodDeckSF', 'OpenPorchSF', 'EnclosedPorch',\n        '3SsnPorch', 'ScreenPorch', 'PoolArea', 'MiscVal', 'MoSold',\n        'YrSold']\n    preprocessed[numerical_feat] = preprocessed[numerical_feat].astype(np.float64)\n  \n    # Handling of missing values\n    \n    # Fill missing values with 'NotAvailable' for categorical nominal features (object)\n    nominal_feat = preprocessed.columns[preprocessed.dtypes == np.object].values\n    preprocessed[nominal_feat] = preprocessed[nominal_feat].fillna('NotAvailable')\n    \n    # Fill missing values with -999 for categorical ordinal features (int64)\n    ordinal_feat = preprocessed.columns[preprocessed.dtypes == np.int64].values\n    preprocessed[ordinal_feat] = preprocessed[ordinal_feat].fillna(-999)\n    \n    # Fill missing values with the mean for numerical features (float64)\n    numerical_feat = preprocessed.columns[preprocessed.dtypes == np.float64].values\n    preprocessed[numerical_feat] = preprocessed[numerical_feat].fillna(\n        preprocessed[numerical_feat].mean())\n    \n    return preprocessed","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"da1222f9beca9cbfeb2ef6915e4a436dbab1e372"},"cell_type":"code","source":"# Preprocess the data and prepare the X_train & y_train dataframes\ntrain_preprocessed = preprocess_features(train)\nX_train = train_preprocessed.drop(['SalePrice', 'Id'], axis=1)\ny_train = train_preprocessed['SalePrice']\n\n# Apply the same preprocessing to the X_test dataframe\ntest_preprocessed = preprocess_features(test)\nX_test = test_preprocessed.drop(['Id'], axis=1)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"c0d171b3e1b7b385ecad8001f8120b851ffbb99f"},"cell_type":"markdown","source":"# Exploratory Data Analysis"},{"metadata":{"_uuid":"85adefb36aeda9011deae32a309cfb3ce680b27e"},"cell_type":"markdown","source":"## Correlation with target variable\nLet's start by visualizing the correlation between the numerical features and the target variable (SalePrice). The heatmap shows that some of the features that have high correlation with the target variable also have high correlation betwen themselves (ie. TotRmsAbvGrd and GrLivArea). To avoid multicolinearity problems during regression analysis (https://en.wikipedia.org/wiki/Multicollinearity) only one of those features should be included in the model."},{"metadata":{"trusted":true,"_uuid":"06794c06c382c53888618d3e82b8ee1f25fcf70d"},"cell_type":"code","source":"# Visualize the correlation of numerical features with the target\nfig, ax = plt.subplots(1, 1, figsize=(12, 12))\ncolumn_filter = (train_preprocessed.dtypes == np.float64) | \\\n                (train_preprocessed.columns == 'SalePrice')\nnumerical_feat = train_preprocessed.columns[column_filter].values\ncorr = train_preprocessed[numerical_feat].corr()\nsns.heatmap(corr, cmap='YlGnBu', ax=ax)\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"17750093f76255de0d80d1c160f628bfb6c78cec"},"cell_type":"markdown","source":"## Pair plot of features with highest correlation\n\nThe pair plot below allows us to visualize the relationship between the numerical features and the target variable using a scatter plot. It is also possible to visualize the distribution of the numerical features in the diagonal. Some outliers seem to be present (top right in the scatter plots**)."},{"metadata":{"trusted":true,"_uuid":"0b88b6ba8ee22031b079282a2016b506c95e6ed7"},"cell_type":"code","source":"# Visualize the relationship of the features with highest correlation to the target variable\nhigh_corr_features = corr.index[corr['SalePrice'] > .52].values\ng = sns.pairplot(train[high_corr_features])\ng.fig.set_size_inches(12, 10)\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"dba0c99bbe2f9bc2beecd5adceee914ad459d7b8"},"cell_type":"markdown","source":"## Point plot of nominal and ordinal features\n\nLet's verifiy the relationship between nominal and ordinal features and the target variable."},{"metadata":{"trusted":true,"_uuid":"985caef7b3c0af12923b5ea3c909ab5d4d19b378"},"cell_type":"code","source":"# Visualize the \"SalePrice\" by \"Neighborhood\"\nplot_pointplot(train_preprocessed, 'SalePrice', \n               ['Neighborhood'], height=5, aspect=2.4)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"3e55c561ff98e776f9f827b228dae9dee0553155"},"cell_type":"code","source":"# Visualize the \"SalePrice\" by the rest of the categorical features\ncolumn_filter = ((train_preprocessed.dtypes == np.object) | \\\n                (train_preprocessed.dtypes == np.int64)) & \\\n                (train_preprocessed.columns != 'SalePrice') & \\\n                (train_preprocessed.columns != 'Id')\ncategorical_feat = train_preprocessed.columns[column_filter].values\nremaining_cat_features = list(set(categorical_feat) - set(['Neighborhood']))\nsplits = np.array_split(np.array(remaining_cat_features), \n                        len(remaining_cat_features)//4)\nfor cat_features in splits:\n    plot_pointplot(train_preprocessed, 'SalePrice', \n                   cat_features.tolist(), height=4, aspect=.6)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"585dc70d942320f75fddf227286ce0a3df85607e"},"cell_type":"markdown","source":"## Distribution of the target variable (SalePrice)\n\nThe SalePrice distribution is skewed to the right (positive skewness) a log transformation could help reduce the skew."},{"metadata":{"trusted":true,"_uuid":"6d70d44f901037a5c1268fb472a6ca8a429b3023"},"cell_type":"code","source":"# Verify the shape of the target feature\nfig, ax = plt.subplots(1, 1, figsize=(12, 5))\nax.set_title('SalePrice')\nsns.distplot(train_preprocessed['SalePrice'], hist=False, ax=ax)\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"ce39796c2239eebc5d8b77fa555c501fb7bdec02"},"cell_type":"markdown","source":"# Feature Selection\n\nLet's use a decision tree to identify the features with highest importance. The results from the decision tree correspond overall to the features with  highest correlation identified before."},{"metadata":{"trusted":true,"_uuid":"a5425fe73cbc1a429abebb0697e13584a6198e2b"},"cell_type":"code","source":"# Nominal features need to be converted to ordinal for use in a \n# decision tree regressor\nnominal_feat = X_train.columns[X_train.dtypes == np.object].values\nrest_of_feat = np.array(list(set(X_train.columns) - set(nominal_feat)))\nall_feat = np.concatenate((rest_of_feat, nominal_feat))\n\n# Transform the nominal features\ncolumn_transf = ColumnTransformer([\n    ('other_feat', 'passthrough', rest_of_feat),\n    ('nominal_feat', OrdinalEncoder(), nominal_feat)])\nX_train_transf = column_transf.fit_transform(X_train)\n\n# Fit the decision tree regressor\nestimator = DecisionTreeRegressor(criterion='friedman_mse')\nestimator.fit(X_train_transf, y_train)\n\n# Features with the highest importances\nix_top_feat = np.argsort(estimator.feature_importances_)[-20:]\ntop_feature_names = all_feat[ix_top_feat]\ntop_feature_imp = estimator.feature_importances_[ix_top_feat]\n\n# Plot the top feature importances\nfig, ax = plt.subplots(1, 1, figsize=(12, 10))\nplt.barh(np.arange(len(top_feature_imp)), top_feature_imp, align='center')\nplt.yticks(np.arange(len(top_feature_imp)), top_feature_names)\nplt.xlabel('Feature importance')\nplt.ylabel('Feature')\nplt.ylim(-1, len(top_feature_imp))\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"52015c9d1de6a3f2801c6edc2f90ebc8f43d08a8"},"cell_type":"markdown","source":"# Model Selection\n\nNow that we have cleaned up the data and explored its characteristics we can compare the performance of different models in order to select the best. The performance of each model is evaluated using a 5x2 nested cross-validation (https://sebastianraschka.com/faq/docs/evaluate-a-model.html)."},{"metadata":{"_uuid":"05f651c06e642cd5804a0b17af61f1cd4788ea19"},"cell_type":"markdown","source":"## Linear Regression\n\nLet's start with a linear regression model that applies L2 regularization (Ridge). The numerical features are binned using a QuantileTransformer and the categorical features are one-hot encoded. A log transformation is applied to the target variable to reduce the skew. Polynomial features are also added to test non-linear relationships with the target variable."},{"metadata":{"trusted":true,"_uuid":"1db92cce9966969f5e398110633a910d6c928c77"},"cell_type":"code","source":"# Linear Regression\n\ndef create_linear_feature_transformations(housing_data):\n    # Features with highest importance\n    selected_features = [\n        'YearRemodAdd', 'WoodDeckSF', 'YearBuilt', 'MoSold', 'GarageType',\n        'KitchenAbvGr', 'MasVnrType', 'CentralAir', 'BsmtUnfSF', 'LotArea',\n        'GarageArea', 'LotFrontage', 'GarageCars', 'Neighborhood',\n        'BsmtFinSF1', '1stFlrSF', 'TotalBsmtSF', '2ndFlrSF', 'GrLivArea',\n        'OverallQual']\n\n    numerical_feat = housing_data.columns[\n        (housing_data.dtypes == np.float64) & \n        (housing_data.columns.isin(selected_features))].values\n    ordinal_feat = housing_data.columns[\n        (housing_data.dtypes == np.int64) &\n        (housing_data.columns.isin(selected_features))].values\n    nominal_feat = housing_data.columns[\n        (housing_data.dtypes == np.object) &\n        (housing_data.columns.isin(selected_features))].values\n    \n    # Column transformations\n    column_transf = ColumnTransformer([\n        ('numeric_feat', QuantileTransformer(n_quantiles=10), numerical_feat),\n        ('ordinal_feat', OneHotEncoder(sparse=False), ordinal_feat),\n        ('nominal_feat', OneHotEncoder(sparse=False), nominal_feat)])\n    \n    # Polynomials and interactions\n    poly_transf = PolynomialFeatures(degree=2)\n    \n    p = Pipeline([\n            ('column_transf', column_transf),\n            ('poly_transf', poly_transf)])    \n    \n    return p\n\n# Apply transformations to the training dataset\ntransf = create_linear_feature_transformations(X_train)\nX_train_transf = transf.fit_transform(X_train)\n\n# Evaluate the linear model\nestimator = Ridge()\nparam_grid = {'alpha': [.1, 1, 10]}\nscoring = {'r2': 'r2', \n           'rmse': rmse_scorer}\nnested_cross_validation(estimator, param_grid, \n                        X_train_transf, np.log(y_train), scoring);","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"5e8bed1ec292dca1163313f0c9d27e51f8dc5efe"},"cell_type":"markdown","source":"## Support Vector Regression\n\nNext, let's try a support vector regression (SVR) using a \"rbf\" kernel to verify if a non-linear model produces better results than a linear regression model. The numerical and ordinal features are scaled using a MinMaxScaler and the nominal features are one-hot encoded. A log transformation is applied to the target variable to reduce the skew."},{"metadata":{"trusted":true,"_uuid":"60da99686accf6ec31cdb1f07ee4afecc266df9e"},"cell_type":"code","source":"# SVR\n\ndef create_svm_feature_transformations(housing_data):\n    # Features with highest importance\n    selected_features = [\n        'YearRemodAdd', 'WoodDeckSF', 'YearBuilt', 'MoSold', 'GarageType',\n        'KitchenAbvGr', 'MasVnrType', 'CentralAir', 'BsmtUnfSF', 'LotArea',\n        'GarageArea', 'LotFrontage', 'GarageCars', 'Neighborhood',\n        'BsmtFinSF1', '1stFlrSF', 'TotalBsmtSF', '2ndFlrSF', 'GrLivArea',\n        'OverallQual']\n    \n    numerical_feat = housing_data.columns[\n        (housing_data.dtypes == np.float64) & \n        (housing_data.columns.isin(selected_features))].values\n    ordinal_feat = housing_data.columns[\n        (housing_data.dtypes == np.int64) &\n        (housing_data.columns.isin(selected_features))].values\n    nominal_feat = housing_data.columns[\n        (housing_data.dtypes == np.object) &\n        (housing_data.columns.isin(selected_features))].values\n    \n    # Column transformations\n    column_transf = ColumnTransformer([\n        ('numeric_feat', MinMaxScaler(), numerical_feat),\n        ('ordinal_feat', MinMaxScaler(), ordinal_feat),\n        ('nominal_feat', OneHotEncoder(sparse=False), nominal_feat)])\n    \n    p = Pipeline([('column_transf', column_transf)])    \n    \n    return p\n\n# Apply transformations to the training dataset\ntransf = create_svm_feature_transformations(X_train)\nX_train_transf = transf.fit_transform(X_train)\n\n# Evaluate the svm model\nestimator = SVR(kernel='rbf')\nparam_grid = {'C': np.logspace(-2, 8, 5), 'gamma': ['auto', 'scale']}\nscoring = {'r2': 'r2', \n           'rmse': rmse_scorer}\nnested_cross_validation(estimator, param_grid, \n                        X_train_transf, np.log(y_train), scoring);","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"b144f070ead235e42d1cf869ff6d01e607bad146"},"cell_type":"markdown","source":"## Gradient Boosting Tree\nNext, let's try a gradient boosting tree (xgboost). There is no need to transform the numerical and ordinal features. The nominal features are encoded using an OrdinalEncoder."},{"metadata":{"trusted":true,"_uuid":"03ca72c70f997a4d89c3bc23655c38a164b42829"},"cell_type":"code","source":"# Gradient Boosting Tree\n\ndef create_tree_feature_transformations(housing_data):\n    # Take all the features\n    selected_features = housing_data.columns\n\n    numerical_feat = housing_data.columns[\n        (housing_data.dtypes == np.float64) & \n        (housing_data.columns.isin(selected_features))].values\n    ordinal_feat = housing_data.columns[\n        (housing_data.dtypes == np.int64) &\n        (housing_data.columns.isin(selected_features))].values\n    nominal_feat = housing_data.columns[\n        (housing_data.dtypes == np.object) &\n        (housing_data.columns.isin(selected_features))].values\n    \n    # Column transformations\n    column_transf = ColumnTransformer([\n        ('numeric_feat', 'passthrough', numerical_feat),\n        ('ordinal_feat', 'passthrough', ordinal_feat),\n        ('nominal_feat', OrdinalEncoder(), nominal_feat)])\n    \n    p = Pipeline([('column_transf', column_transf)])    \n    \n    return p\n\n# Apply transformations to the training dataset\ntree_transf = create_tree_feature_transformations(X_train)\nX_train_transf = tree_transf.fit_transform(X_train)\n\n# Evaluate xgboost model\nestimator = xgb.XGBRegressor(booster='gbtree', \n                             objective='reg:linear', \n                             n_estimators=100)\nparam_grid = {'lambda': np.logspace(-3, 3, 3), \n              'alpha': np.logspace(-3, 3, 3), \n              'max_depth': [3, 6, 10]}\nscoring = {'r2': 'r2', \n           'rmse': rmse_scorer}\nnested_cross_validation(estimator, param_grid, \n                        X_train_transf, np.log(y_train), scoring);","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"17e7fbd608950227c183f3276a85517ecd7f07e6"},"cell_type":"markdown","source":"# Model Training & Fine Tuning\nThe gradient boosting tree showed the best performance. Let's train it and fine tune it to improve the result obtained during the model selection step."},{"metadata":{"trusted":true,"_uuid":"603a22b860de9dd386c1899d37c2dacb8c9e04be"},"cell_type":"code","source":"# Fine tune the gradient boosting tree\n\n# Apply transformations to the training dataset\ntree_transf = create_tree_feature_transformations(X_train)\ntree_transf.fit(\n    pd.concat([X_train, X_test])) # Need to fit on X_test also so that\n                                  # there are no 'unknown' categories when \n                                  # doing the predictions\nX_train_transf = tree_transf.transform(X_train)\n\n# Evaluate xgboost model\nestimator = xgb.XGBRegressor(booster='gbtree', \n                             objective='reg:linear',\n                             random_state=0)\nparam_grid = {'lambda': [0, 1], \n              'alpha': [0, 1], \n              'max_depth': [1, 3, 10],\n              'eta': [0, .01],\n              'colsample_bytree': [.01, .1, 1],\n              'n_estimators': [300]}\ngs = GridSearchCV(estimator, param_grid, cv=10, \n                  scoring='r2', n_jobs=-1)\ngs.fit(X_train_transf, np.log(y_train))\n\nprint('Best %s score: %.4f' % ('r2', gs.best_score_))\nprint('Best params: %s' % (gs.best_params_))","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"6e913fcfda6caea107c23c7e0145db6cba935af6"},"cell_type":"markdown","source":"# Predictions & Submission File\nThe final step is to predict the house prices using the fine-tuned model and create the submission file."},{"metadata":{"trusted":true,"_uuid":"83a3f84ce85a6b4fc2a1be199650eb3a78dd21db"},"cell_type":"code","source":"# Prepare the final submission\nX_test_transf = tree_transf.transform(X_test)\n\n# Evaluate xgboost model\nestimator = gs.best_estimator_\ny_pred = estimator.predict(X_test_transf)\n\n# Create the submission file\nsubmission = pd.DataFrame({'Id': test_preprocessed.Id, \n                           'SalePrice': np.exp(y_pred)}) # Reverse the log transformation\nsubmission.to_csv('submission.csv', index=False)","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}