{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-12T09:45:56.738569Z","iopub.execute_input":"2022-07-12T09:45:56.739973Z","iopub.status.idle":"2022-07-12T09:45:56.763277Z","shell.execute_reply.started":"2022-07-12T09:45:56.739861Z","shell.execute_reply":"2022-07-12T09:45:56.761906Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data prep.\n\nWe start by loading the data into a pandas.DataFrame with the read_csv() function, and then summarising the data with the `.info()` method. ","metadata":{}},{"cell_type":"code","source":"train_data = pd.read_csv(\"/kaggle/input/house-prices-advanced-regression-techniques/train.csv\")\ntrain_data.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-12T09:45:56.775075Z","iopub.execute_input":"2022-07-12T09:45:56.776349Z","iopub.status.idle":"2022-07-12T09:45:56.870830Z","shell.execute_reply.started":"2022-07-12T09:45:56.776312Z","shell.execute_reply":"2022-07-12T09:45:56.869861Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We can see from the output above that there are a large number of features (80), and that many of these are categorical. Before deciding what features to use we the `get_dummies()` function to encode the categorical features.","metadata":{}},{"cell_type":"code","source":"train_data_encoded = pd.get_dummies(train_data)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T09:45:56.874840Z","iopub.execute_input":"2022-07-12T09:45:56.875583Z","iopub.status.idle":"2022-07-12T09:45:56.931387Z","shell.execute_reply.started":"2022-07-12T09:45:56.875545Z","shell.execute_reply":"2022-07-12T09:45:56.930526Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We then plot Pearson correlation coefficient for the 100 features that have the highest correlation with the `SalePrice`.","metadata":{}},{"cell_type":"code","source":"train_data_encoded.corr()['SalePrice'].iloc[:-1].abs().sort_values()[::-1][1:101].plot.bar(figsize=(25,4))","metadata":{"execution":{"iopub.status.busy":"2022-07-12T09:45:56.933886Z","iopub.execute_input":"2022-07-12T09:45:56.934788Z","iopub.status.idle":"2022-07-12T09:45:59.178061Z","shell.execute_reply.started":"2022-07-12T09:45:56.934754Z","shell.execute_reply":"2022-07-12T09:45:59.176675Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We will use these 100 most correlated features to train our model.","metadata":{}},{"cell_type":"code","source":"X = train_data_encoded[train_data_encoded.corr()['SalePrice'].iloc[:-1].abs().sort_values()[::-1][1:101].index.tolist()]\ny = train_data_encoded['SalePrice']","metadata":{"execution":{"iopub.status.busy":"2022-07-12T09:45:59.179591Z","iopub.execute_input":"2022-07-12T09:45:59.179937Z","iopub.status.idle":"2022-07-12T09:45:59.534556Z","shell.execute_reply.started":"2022-07-12T09:45:59.179906Z","shell.execute_reply":"2022-07-12T09:45:59.533323Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Before doing an training we need to replace any missing values. We do this by simply replacing with the mean.","metadata":{}},{"cell_type":"code","source":"from sklearn.impute import SimpleImputer\n\nimp_mean = SimpleImputer(missing_values=np.nan, strategy='mean')\nX_ImpMean = imp_mean.fit_transform(X)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T09:45:59.537003Z","iopub.execute_input":"2022-07-12T09:45:59.537435Z","iopub.status.idle":"2022-07-12T09:46:00.247077Z","shell.execute_reply.started":"2022-07-12T09:45:59.537402Z","shell.execute_reply":"2022-07-12T09:46:00.245895Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Model training","metadata":{}},{"cell_type":"code","source":"from sklearn.metrics import mean_squared_error","metadata":{"execution":{"iopub.status.busy":"2022-07-12T09:46:00.248732Z","iopub.execute_input":"2022-07-12T09:46:00.249554Z","iopub.status.idle":"2022-07-12T09:46:00.256546Z","shell.execute_reply.started":"2022-07-12T09:46:00.249496Z","shell.execute_reply":"2022-07-12T09:46:00.254346Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import MinMaxScaler, QuantileTransformer\nfrom sklearn.gaussian_process import GaussianProcessRegressor\nfrom sklearn.gaussian_process.kernels import ConstantKernel, RBF, Matern, DotProduct\nfrom sklearn.model_selection import train_test_split","metadata":{"execution":{"iopub.status.busy":"2022-07-12T09:46:00.258258Z","iopub.execute_input":"2022-07-12T09:46:00.258956Z","iopub.status.idle":"2022-07-12T09:46:00.273888Z","shell.execute_reply.started":"2022-07-12T09:46:00.258910Z","shell.execute_reply":"2022-07-12T09:46:00.272938Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train, X_val, y_train, y_val = train_test_split(X_ImpMean, y, test_size=0.2, random_state=1)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T09:46:00.276282Z","iopub.execute_input":"2022-07-12T09:46:00.277021Z","iopub.status.idle":"2022-07-12T09:46:00.296787Z","shell.execute_reply.started":"2022-07-12T09:46:00.276974Z","shell.execute_reply":"2022-07-12T09:46:00.295456Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"xscaler = QuantileTransformer(output_distribution='normal')\nyscaler = QuantileTransformer(output_distribution='normal')","metadata":{"execution":{"iopub.status.busy":"2022-07-12T09:46:00.298825Z","iopub.execute_input":"2022-07-12T09:46:00.300246Z","iopub.status.idle":"2022-07-12T09:46:00.305626Z","shell.execute_reply.started":"2022-07-12T09:46:00.300199Z","shell.execute_reply":"2022-07-12T09:46:00.304527Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train_scaled = xscaler.fit_transform(X_train)\ny_train_scaled = yscaler.fit_transform(np.array(y_train).reshape(-1,1))","metadata":{"execution":{"iopub.status.busy":"2022-07-12T09:46:00.307384Z","iopub.execute_input":"2022-07-12T09:46:00.308108Z","iopub.status.idle":"2022-07-12T09:46:00.585053Z","shell.execute_reply.started":"2022-07-12T09:46:00.308063Z","shell.execute_reply":"2022-07-12T09:46:00.583566Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"kernel = ConstantKernel()+Matern()+DotProduct()\ngpr = GaussianProcessRegressor(kernel=kernel, n_restarts_optimizer=5, normalize_y=True, random_state=1)\ngpr.fit(X_train_scaled, y_train_scaled);","metadata":{"execution":{"iopub.status.busy":"2022-07-12T09:46:00.587125Z","iopub.execute_input":"2022-07-12T09:46:00.587805Z","iopub.status.idle":"2022-07-12T09:46:58.110645Z","shell.execute_reply.started":"2022-07-12T09:46:00.587746Z","shell.execute_reply":"2022-07-12T09:46:58.108863Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds_val = yscaler.inverse_transform(gpr.predict(xscaler.transform(X_val))).reshape(-1,)\nmean_squared_error(y_val, preds_val, squared=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T09:46:58.115605Z","iopub.execute_input":"2022-07-12T09:46:58.120789Z","iopub.status.idle":"2022-07-12T09:46:58.291091Z","shell.execute_reply.started":"2022-07-12T09:46:58.120713Z","shell.execute_reply":"2022-07-12T09:46:58.289740Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data = pd.read_csv(\"/kaggle/input/house-prices-advanced-regression-techniques/test.csv\")\ntest_data_encoded = pd.get_dummies(test_data)\n\nX_test = test_data_encoded[train_data_encoded.corr()['SalePrice'].iloc[:-1].abs().sort_values()[::-1][1:101].index.tolist()]\nX_test_ImpMean = imp_mean.transform(X_test)\nX_test_ImpMean_scaled = xscaler.transform(X_test_ImpMean)\n\noutput = pd.DataFrame({'Id': test_data.Id,\n                       'SalePrice': yscaler.inverse_transform(gpr.predict(X_test_ImpMean_scaled)).reshape(-1,)})\noutput.to_csv('submission.csv', index=False)\nprint(\"Your submission was successfully saved!\")","metadata":{"execution":{"iopub.status.busy":"2022-07-12T09:46:58.293202Z","iopub.execute_input":"2022-07-12T09:46:58.293666Z","iopub.status.idle":"2022-07-12T09:46:59.103692Z","shell.execute_reply.started":"2022-07-12T09:46:58.293620Z","shell.execute_reply":"2022-07-12T09:46:59.102376Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}