{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np \nimport pandas as pd \n\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.preprocessing import OneHotEncoder, StandardScaler\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.compose import ColumnTransformer\nfrom sklearn.linear_model import LinearRegression\nfrom sklearn.metrics import mean_squared_error","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-06T11:31:15.967337Z","iopub.execute_input":"2022-08-06T11:31:15.967773Z","iopub.status.idle":"2022-08-06T11:31:15.973776Z","shell.execute_reply.started":"2022-08-06T11:31:15.967740Z","shell.execute_reply":"2022-08-06T11:31:15.972883Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv('../input/house-prices-advanced-regression-techniques/train.csv', index_col=0)\ntest = pd.read_csv('../input/house-prices-advanced-regression-techniques/test.csv', index_col=0)\ntrain","metadata":{"execution":{"iopub.status.busy":"2022-08-06T11:54:20.022534Z","iopub.execute_input":"2022-08-06T11:54:20.022987Z","iopub.status.idle":"2022-08-06T11:54:20.113406Z","shell.execute_reply.started":"2022-08-06T11:54:20.022946Z","shell.execute_reply":"2022-08-06T11:54:20.112420Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# X_train, X_test, y_train, y_test = train_test_split(train.drop(columns='SalePrice'), train['SalePrice'], test_size=0.1) #implementation before submission\n\nX_train, y_train = train.drop(columns='SalePrice'), train['SalePrice']\nX_test = test\n\ncat_cols = [col for col in X_train.columns if X_train[col].dtype=='object'] #Categorical Features\nnum_cols = [col for col in X_train.columns if X_train[col].dtype!='object'] #Numeric Features","metadata":{"execution":{"iopub.status.busy":"2022-08-06T11:43:45.741096Z","iopub.execute_input":"2022-08-06T11:43:45.741594Z","iopub.status.idle":"2022-08-06T11:43:45.761867Z","shell.execute_reply.started":"2022-08-06T11:43:45.741553Z","shell.execute_reply":"2022-08-06T11:43:45.760844Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We work with categorical feature seperately from the numeric features as there are some things which need to be done differently for numeric and categoric features.","metadata":{}},{"cell_type":"markdown","source":"The following code uses Pipelines and ColumnTransformer. These tasks can be done without using these functions, but they save us the headache of doing all the steps again for test sets. If you feel uncomfortable dealing with them, I suggest you go through the [intermediate machine learning](https://www.kaggle.com/learn/intermediate-machine-learning) course in kaggle (specifically lesson 4 of the course).","metadata":{}},{"cell_type":"code","source":"numeric_transformer = Pipeline(steps=[\n    ('imputer', SimpleImputer(strategy='mean')), #replace numeric missing values by mean\n    ('scaler', StandardScaler())\n])\n\ncategorical_transformer = Pipeline(steps=[\n    ('imputer', SimpleImputer(strategy='constant', fill_value='NA')), #replace categorical missing values by Not Available.\n    ('encoder', OneHotEncoder(handle_unknown='ignore'))\n])\n\npreprocessor = ColumnTransformer(transformers=[\n    ('numeric', numeric_transformer, num_cols),\n    ('categoric', categorical_transformer, cat_cols)\n])\n\npipe = Pipeline(steps=[\n    ('preprocess', preprocessor),\n    ('model', LinearRegression())\n])\n\npipe.fit(X_train, y_train);","metadata":{"execution":{"iopub.status.busy":"2022-08-06T12:06:35.833495Z","iopub.execute_input":"2022-08-06T12:06:35.833899Z","iopub.status.idle":"2022-08-06T12:06:36.427040Z","shell.execute_reply.started":"2022-08-06T12:06:35.833867Z","shell.execute_reply":"2022-08-06T12:06:36.425903Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The rest is predicting and submitting.","metadata":{}},{"cell_type":"code","source":"pd.Series(pipe.predict(X_test), index=X_test.index, name='SalePrice').to_csv('submission.csv', index_label='Id')","metadata":{"execution":{"iopub.status.busy":"2022-08-06T11:50:13.738722Z","iopub.execute_input":"2022-08-06T11:50:13.739191Z","iopub.status.idle":"2022-08-06T11:50:13.787316Z","shell.execute_reply.started":"2022-08-06T11:50:13.739152Z","shell.execute_reply":"2022-08-06T11:50:13.785897Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.read_csv('./submission.csv')","metadata":{"execution":{"iopub.status.busy":"2022-08-06T11:50:15.340271Z","iopub.execute_input":"2022-08-06T11:50:15.340725Z","iopub.status.idle":"2022-08-06T11:50:15.357039Z","shell.execute_reply.started":"2022-08-06T11:50:15.340687Z","shell.execute_reply":"2022-08-06T11:50:15.356148Z"},"trusted":true},"execution_count":null,"outputs":[]}]}