{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Setup","metadata":{}},{"cell_type":"markdown","source":"First, we import the Python librairies that will we need for this project.","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nfrom sklearn.model_selection import train_test_split","metadata":{"execution":{"iopub.status.busy":"2022-07-29T10:53:22.716884Z","iopub.execute_input":"2022-07-29T10:53:22.719669Z","iopub.status.idle":"2022-07-29T10:53:22.724719Z","shell.execute_reply.started":"2022-07-29T10:53:22.719606Z","shell.execute_reply":"2022-07-29T10:53:22.723248Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We can then read the data, remove rows that have no price and split the dataset in a training set and a validation one.","metadata":{}},{"cell_type":"code","source":"# Read the data\nX = pd.read_csv('../input/house-prices-advanced-regression-techniques/train.csv', index_col='Id')\nX_test_full = pd.read_csv('../input/house-prices-advanced-regression-techniques/test.csv', index_col='Id')\n\n# Remove rows with missing target, separate target from predictors\nX.dropna(axis=0, subset=['SalePrice'], inplace=True)\ny = X.SalePrice              \nX.drop(['SalePrice'], axis=1, inplace=True)\n\n# Break off validation set from training data\nX_train_full, X_valid_full, y_train, y_valid = train_test_split(X, y, train_size=0.8, test_size=0.2,\n                                                                random_state=0)","metadata":{"execution":{"iopub.status.busy":"2022-07-29T10:53:22.784687Z","iopub.execute_input":"2022-07-29T10:53:22.785105Z","iopub.status.idle":"2022-07-29T10:53:22.859927Z","shell.execute_reply.started":"2022-07-29T10:53:22.785074Z","shell.execute_reply":"2022-07-29T10:53:22.858605Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We recover the categorical columns with a low cardinality and the numerical columns.","metadata":{}},{"cell_type":"code","source":"# \"Cardinality\" means the number of unique values in a column\n# Select categorical columns with relatively low cardinality (convenient but arbitrary)\ncategorical_cols = [cname for cname in X_train_full.columns if X_train_full[cname].nunique() < 10 and \n                        X_train_full[cname].dtype == \"object\"]\n\n# Select numeric columns\nnumerical_cols = [cname for cname in X_train_full.columns if X_train_full[cname].dtype in ['int64', 'float64']]\n\n# Keep selected columns only\nmy_cols = categorical_cols + numerical_cols","metadata":{"execution":{"iopub.status.busy":"2022-07-29T10:53:22.879564Z","iopub.execute_input":"2022-07-29T10:53:22.880030Z","iopub.status.idle":"2022-07-29T10:53:22.906412Z","shell.execute_reply.started":"2022-07-29T10:53:22.879992Z","shell.execute_reply":"2022-07-29T10:53:22.905077Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We make copies of the training and validation datasets to avoid modifing the original and only use the columns that we want.","metadata":{}},{"cell_type":"code","source":"X_train = X_train_full[my_cols].copy()\nX_valid = X_valid_full[my_cols].copy()\nX_test = X_test_full[my_cols].copy()","metadata":{"execution":{"iopub.status.busy":"2022-07-29T10:53:22.930772Z","iopub.execute_input":"2022-07-29T10:53:22.931249Z","iopub.status.idle":"2022-07-29T10:53:22.946933Z","shell.execute_reply.started":"2022-07-29T10:53:22.931211Z","shell.execute_reply":"2022-07-29T10:53:22.945808Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Creating the pipeline","metadata":{}},{"cell_type":"code","source":"from sklearn.compose import ColumnTransformer\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.preprocessing import OneHotEncoder\nfrom xgboost import XGBRegressor\nfrom sklearn.metrics import mean_absolute_error\n\n# Preprocessing for numerical data\nnumerical_transformer = SimpleImputer(strategy='median')\n\n# Preprocessing for categorical data\ncategorical_transformer = Pipeline(steps=[\n    ('imputer', SimpleImputer(strategy='most_frequent')),\n    ('onehot', OneHotEncoder(handle_unknown='ignore'))\n])\n\n# Bundle preprocessing for numerical and categorical data\npreprocessor = ColumnTransformer(\n    transformers=[\n        ('num', numerical_transformer, numerical_cols),\n        ('cat', categorical_transformer, categorical_cols)\n    ])\n\n# Define model\nmodel = XGBRegressor(random_state=0, n_estimators=1000, learning_rate=0.05)\n\n# Bundle preprocessing and modeling code in a pipeline\nclf = Pipeline(steps=[('preprocessor', preprocessor),\n                      ('model', model)\n                     ])\n\n# Preprocessing of training data, fit model \nclf.fit(X_train, y_train)\n\n# Preprocessing of validation data, get predictions\npreds = clf.predict(X_valid)\n\nprint('MAE:', mean_absolute_error(y_valid, preds))","metadata":{"execution":{"iopub.status.busy":"2022-07-29T10:56:25.676241Z","iopub.execute_input":"2022-07-29T10:56:25.676630Z","iopub.status.idle":"2022-07-29T10:56:25.700570Z","shell.execute_reply.started":"2022-07-29T10:56:25.676601Z","shell.execute_reply":"2022-07-29T10:56:25.699148Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Submitting the results","metadata":{}},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"# Preprocessing of test data, fit model\npreds_test = clf.predict(X_test)\n\n# Save test predictions to file\noutput = pd.DataFrame({'Id': X_test.index,\n                       'SalePrice': preds_test})\noutput.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-29T10:57:05.307240Z","iopub.execute_input":"2022-07-29T10:57:05.307919Z","iopub.status.idle":"2022-07-29T10:57:05.448799Z","shell.execute_reply.started":"2022-07-29T10:57:05.307869Z","shell.execute_reply":"2022-07-29T10:57:05.447897Z"},"trusted":true},"execution_count":null,"outputs":[]}]}