{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## Hello everybody, I'm in my first month in learning ML, and this might be my first ML project,so I've been reading some other projects on kaggle that inspired me to preform this project as it's now, those projects that inspired me were:\n### [Prices Prediction - EDA + XGBoost Regression](https://www.kaggle.com/code/vitorgamalemos/prices-prediction-eda-xgboost-regression)\n### [Intro & Intermediate Machine Learning](https://www.kaggle.com/code/luan1krk/intro-intermediate-machine-learning)\n### [k-NN regression clearly explained](https://www.kaggle.com/code/leodaniel/k-nn-regression-clearly-explained)\n### [🏠 Housing Prices: Pipelines + Custom Transformer](https://www.kaggle.com/code/wojteksy/housing-prices-pipelines-custom-transformer)\n### [House Prices - Regression Prediction](https://www.kaggle.com/code/dalao1002/house-prices-regression-prediction)\n\n## So let's start exploring what I've learned so far!","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19"}},{"cell_type":"markdown","source":"## Let's start with importing the needed libraries!","metadata":{}},{"cell_type":"code","source":"import numpy as np \nimport pandas as pd \npd.set_option('display.max_rows', 100)\nimport matplotlib.pyplot as plt\n%matplotlib inline\nimport seaborn as sns\nsns.set_style('darkgrid')\n\nfrom sklearn.metrics import mean_absolute_error\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.compose import ColumnTransformer\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.impute import KNNImputer\nfrom sklearn.preprocessing import OneHotEncoder\nfrom sklearn.metrics import r2_score\nfrom tqdm import tqdm\nfrom sklearn.preprocessing import StandardScaler\n\nfrom sklearn.ensemble import GradientBoostingRegressor\nfrom xgboost import XGBRegressor\nfrom sklearn.neighbors import KNeighborsRegressor\nfrom sklearn.linear_model import LinearRegression","metadata":{"execution":{"iopub.status.busy":"2022-07-13T18:07:35.808008Z","iopub.execute_input":"2022-07-13T18:07:35.808917Z","iopub.status.idle":"2022-07-13T18:07:37.461581Z","shell.execute_reply.started":"2022-07-13T18:07:35.808821Z","shell.execute_reply":"2022-07-13T18:07:37.460750Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Now let's import and take a quick look on the house prices data!","metadata":{}},{"cell_type":"code","source":"train_full = pd.read_csv('../input/house-prices-advanced-regression-techniques/train.csv', index_col='Id')\ntest_X = pd.read_csv('../input/house-prices-advanced-regression-techniques/test.csv', index_col='Id')\n\ncm = sns.light_palette(\"pink\", as_cmap=True)\ntrain_full.head(20).style.background_gradient(cmap=cm)","metadata":{"execution":{"iopub.status.busy":"2022-07-13T18:07:37.464193Z","iopub.execute_input":"2022-07-13T18:07:37.464720Z","iopub.status.idle":"2022-07-13T18:07:37.694482Z","shell.execute_reply.started":"2022-07-13T18:07:37.464694Z","shell.execute_reply":"2022-07-13T18:07:37.693339Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.DataFrame(train_full.dtypes, columns=[\"type\"]) # Showing the data type for each column in the dataset.","metadata":{"execution":{"iopub.status.busy":"2022-07-13T18:07:37.695617Z","iopub.execute_input":"2022-07-13T18:07:37.695888Z","iopub.status.idle":"2022-07-13T18:07:37.714984Z","shell.execute_reply.started":"2022-07-13T18:07:37.695860Z","shell.execute_reply":"2022-07-13T18:07:37.713415Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## What about a quick statistical look?","metadata":{}},{"cell_type":"code","source":"train_full.describe().style.background_gradient(cmap=cm)","metadata":{"execution":{"iopub.status.busy":"2022-07-13T18:07:37.718598Z","iopub.execute_input":"2022-07-13T18:07:37.718910Z","iopub.status.idle":"2022-07-13T18:07:37.838676Z","shell.execute_reply.started":"2022-07-13T18:07:37.718883Z","shell.execute_reply":"2022-07-13T18:07:37.837606Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Let's see how many null values in each column!","metadata":{}},{"cell_type":"code","source":"isnull = train_full.isnull().sum().sort_values(ascending=False).to_frame()\nisnull.columns = ['How_many']\nisnull['precentage'] = np.around(((isnull / len(train_full) * 100)[(isnull / len(train_full) * 100) != 0]), decimals=2)\nisnull[isnull.How_many > 0].style.background_gradient(cmap=cm)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-13T18:07:37.839900Z","iopub.execute_input":"2022-07-13T18:07:37.840457Z","iopub.status.idle":"2022-07-13T18:07:37.864088Z","shell.execute_reply.started":"2022-07-13T18:07:37.840424Z","shell.execute_reply":"2022-07-13T18:07:37.863165Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Now let's see how the variables correlate with each other!","metadata":{}},{"cell_type":"code","source":"    plt.figure(figsize=(35, 35))\n    sns.heatmap(train_full.corr(), annot=True, cmap=\"YlOrRd\", linewidths=0.1, annot_kws={\"fontsize\":10})\n    plt.title(\"Correlation house prices - return rate\");","metadata":{"execution":{"iopub.status.busy":"2022-07-13T18:07:37.865111Z","iopub.execute_input":"2022-07-13T18:07:37.865410Z","iopub.status.idle":"2022-07-13T18:07:42.386285Z","shell.execute_reply.started":"2022-07-13T18:07:37.865382Z","shell.execute_reply":"2022-07-13T18:07:42.385419Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"correlation = train_full.corr().unstack().sort_values(kind=\"quicksort\", ascending=False)\n\ncorrelation = correlation[correlation!=1]\nprint(\"Top 20 with highest positive correlation\")\nprint(correlation[:20])\nprint(\"Top 20 with highest negative correlation\")\nprint(correlation[-20:][::-1])","metadata":{"execution":{"iopub.status.busy":"2022-07-13T18:07:42.387544Z","iopub.execute_input":"2022-07-13T18:07:42.388011Z","iopub.status.idle":"2022-07-13T18:07:42.405206Z","shell.execute_reply.started":"2022-07-13T18:07:42.387980Z","shell.execute_reply":"2022-07-13T18:07:42.404184Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Now let's get to splitting out data into train and validation sets!","metadata":{}},{"cell_type":"code","source":"train_full.dropna(axis=0,subset = ['SalePrice'], inplace = True)\n\nX_train_full = train_full.drop(['SalePrice'], axis =1)\ny_train_full = train_full.SalePrice\n\nX_train, X_valid, y_train, y_valid = train_test_split(X_train_full,y_train_full,test_size=0.25, random_state=42)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-13T18:07:42.406727Z","iopub.execute_input":"2022-07-13T18:07:42.407005Z","iopub.status.idle":"2022-07-13T18:07:42.434805Z","shell.execute_reply.started":"2022-07-13T18:07:42.406978Z","shell.execute_reply":"2022-07-13T18:07:42.433881Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Geting the names of the categorical columns.\ncat_columns = [column for column in X_train.columns if X_train[column].dtype == 'object']\ncat_columns","metadata":{"execution":{"iopub.status.busy":"2022-07-13T18:07:42.435989Z","iopub.execute_input":"2022-07-13T18:07:42.436301Z","iopub.status.idle":"2022-07-13T18:07:42.449632Z","shell.execute_reply.started":"2022-07-13T18:07:42.436272Z","shell.execute_reply":"2022-07-13T18:07:42.448565Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Geting the names of the numrical columns.\nnum_columns = [column for column in X_train.columns if X_train[column].dtype != 'object']\nnum_columns","metadata":{"execution":{"iopub.status.busy":"2022-07-13T18:07:42.454361Z","iopub.execute_input":"2022-07-13T18:07:42.454881Z","iopub.status.idle":"2022-07-13T18:07:42.462490Z","shell.execute_reply.started":"2022-07-13T18:07:42.454841Z","shell.execute_reply":"2022-07-13T18:07:42.461567Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## It's cleaning and preprocessing data time!","metadata":{}},{"cell_type":"code","source":"# A categorical transformer.\ncat_trans = Pipeline(steps = [\n    ('imputer',SimpleImputer(strategy = 'most_frequent')),\n    ('ohe',OneHotEncoder(handle_unknown = 'ignore'))\n])\n\n# A numrical transformer.\nnum_trans = Pipeline(steps = [\n    ('imputer',KNNImputer(n_neighbors = 5)),\n    ('scaler',StandardScaler())\n])\n\n# A preprocessor that combines the two previous transformers.\npreprocessor = ColumnTransformer(transformers = [\n    ('num', num_trans, num_columns),\n    ('cat', cat_trans, cat_columns)\n],\n    remainder = \"drop\")","metadata":{"execution":{"iopub.status.busy":"2022-07-13T18:07:42.463693Z","iopub.execute_input":"2022-07-13T18:07:42.464108Z","iopub.status.idle":"2022-07-13T18:07:42.477478Z","shell.execute_reply.started":"2022-07-13T18:07:42.464071Z","shell.execute_reply":"2022-07-13T18:07:42.476467Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## After creating our preprocessor, let's get it engaged with our data!","metadata":{}},{"cell_type":"code","source":"preprocessor.fit(X_train)\nX_train_trans = preprocessor.fit_transform(X_train).toarray()\nX_valid_trans = preprocessor.transform(X_valid).toarray()","metadata":{"execution":{"iopub.status.busy":"2022-07-13T18:07:42.478759Z","iopub.execute_input":"2022-07-13T18:07:42.479451Z","iopub.status.idle":"2022-07-13T18:07:42.724197Z","shell.execute_reply.started":"2022-07-13T18:07:42.479413Z","shell.execute_reply":"2022-07-13T18:07:42.723304Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Now it's time for creating models, to accomplish this task I found 4 popular models (algorithms) to do the job, so let's start creating them and then test them to find the best one to solve this task.","metadata":{}},{"cell_type":"markdown","source":"# KNN Regressor","metadata":{}},{"cell_type":"code","source":"results = []\nresults_train = []\npossibles_k = np.arange(3,99, 2)\n\nfor k in tqdm(possibles_k):\n    knn = KNeighborsRegressor(n_neighbors=k)\n    knn.fit(X_train_trans, y_train)\n    y_hat = knn.predict(X_valid_trans)\n    \n    results.append(mean_absolute_error(y_valid, y_hat))\n    \n    y_hat = knn.predict(X_train_trans)\n    results_train.append(mean_absolute_error(y_train, y_hat))\n    \n    \nidx = np.argmin(results)   \nprint(f\"Best k: {possibles_k[idx]}\")\nprint(f\"Best RMSE: {results[idx]}\")","metadata":{"execution":{"iopub.status.busy":"2022-07-13T18:07:42.725640Z","iopub.execute_input":"2022-07-13T18:07:42.726200Z","iopub.status.idle":"2022-07-13T18:07:45.187426Z","shell.execute_reply.started":"2022-07-13T18:07:42.726159Z","shell.execute_reply":"2022-07-13T18:07:45.186565Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"knn_model = KNeighborsRegressor(n_neighbors=9)\nknn_model.fit(X_train_trans,y_train)\ny_pred = knn_model.predict(X_valid_trans)\nmean_absolute_error(y_valid, y_pred)","metadata":{"execution":{"iopub.status.busy":"2022-07-13T18:07:45.191700Z","iopub.execute_input":"2022-07-13T18:07:45.194504Z","iopub.status.idle":"2022-07-13T18:07:45.222073Z","shell.execute_reply.started":"2022-07-13T18:07:45.194468Z","shell.execute_reply":"2022-07-13T18:07:45.221227Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"r2_score(y_valid, y_pred)","metadata":{"execution":{"iopub.status.busy":"2022-07-13T18:07:45.223349Z","iopub.execute_input":"2022-07-13T18:07:45.223841Z","iopub.status.idle":"2022-07-13T18:07:45.231229Z","shell.execute_reply.started":"2022-07-13T18:07:45.223799Z","shell.execute_reply":"2022-07-13T18:07:45.230258Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# XGB Regressor","metadata":{}},{"cell_type":"code","source":"xgb_model = XGBRegressor(\n    n_estimators=1000,\n    max_depth=10,\n    eta=0.1, \n    subsample=0.7, \n    colsample_bytree=0.8\n)\n\nxgb_model = xgb_model.fit(X_train_trans,y_train)\n\npreds_XGB = xgb_model.predict(X_valid_trans)\n\nmean_absolute_error(y_valid, preds_XGB)","metadata":{"execution":{"iopub.status.busy":"2022-07-13T18:07:45.232893Z","iopub.execute_input":"2022-07-13T18:07:45.233637Z","iopub.status.idle":"2022-07-13T18:07:58.730418Z","shell.execute_reply.started":"2022-07-13T18:07:45.233602Z","shell.execute_reply":"2022-07-13T18:07:58.729605Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"r2_score(y_valid, preds_XGB)","metadata":{"execution":{"iopub.status.busy":"2022-07-13T18:07:58.733206Z","iopub.execute_input":"2022-07-13T18:07:58.733655Z","iopub.status.idle":"2022-07-13T18:07:58.740748Z","shell.execute_reply.started":"2022-07-13T18:07:58.733629Z","shell.execute_reply":"2022-07-13T18:07:58.740173Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# GB Regressor","metadata":{}},{"cell_type":"code","source":"GB_model =  GradientBoostingRegressor(random_state=100,\n                                      n_estimators=1000,\n                                      loss='squared_error',\n                                      subsample = 0.34,\n                                      learning_rate = 0.04)\n\nGB_model.fit(X_train_trans, y_train)\n\npreds_GB = GB_model.predict(X_valid_trans)\n\nmean_absolute_error(y_valid, preds_GB)","metadata":{"execution":{"iopub.status.busy":"2022-07-13T18:07:58.741682Z","iopub.execute_input":"2022-07-13T18:07:58.742009Z","iopub.status.idle":"2022-07-13T18:08:01.725091Z","shell.execute_reply.started":"2022-07-13T18:07:58.741988Z","shell.execute_reply":"2022-07-13T18:08:01.723861Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"r2_score(y_valid, preds_GB)","metadata":{"execution":{"iopub.status.busy":"2022-07-13T18:08:01.726127Z","iopub.execute_input":"2022-07-13T18:08:01.726427Z","iopub.status.idle":"2022-07-13T18:08:01.733455Z","shell.execute_reply.started":"2022-07-13T18:08:01.726396Z","shell.execute_reply":"2022-07-13T18:08:01.732585Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Linear Regressor","metadata":{}},{"cell_type":"code","source":"LR_model = LinearRegression()\n\n\nLR_model.fit(X_train_trans, y_train)\npreds_LR = LR_model.predict(X_valid_trans)\n\nmean_absolute_error(y_valid, preds_LR)","metadata":{"execution":{"iopub.status.busy":"2022-07-13T18:08:01.734661Z","iopub.execute_input":"2022-07-13T18:08:01.735512Z","iopub.status.idle":"2022-07-13T18:08:01.833483Z","shell.execute_reply.started":"2022-07-13T18:08:01.735480Z","shell.execute_reply":"2022-07-13T18:08:01.832722Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"r2_score(y_valid, preds_LR)","metadata":{"execution":{"iopub.status.busy":"2022-07-13T18:09:16.875421Z","iopub.execute_input":"2022-07-13T18:09:16.875735Z","iopub.status.idle":"2022-07-13T18:09:16.882919Z","shell.execute_reply.started":"2022-07-13T18:09:16.875704Z","shell.execute_reply":"2022-07-13T18:09:16.882063Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Of the four models, the best and most efficient was one of the simplest, I mean the regular gradient boost regressor.","metadata":{}},{"cell_type":"markdown","source":"## Now before creating my submission I'll fit the preprocessor and the GBRegressor model with the train_full dataset so it gets more efficient in predicting the prices.","metadata":{}},{"cell_type":"code","source":"preprocessor.fit(X_train_full)\nX_train_full_trans = preprocessor.fit_transform(X_train_full).toarray()\ntest_X_trans = preprocessor.transform(test_X).toarray()","metadata":{"execution":{"iopub.status.busy":"2022-07-13T18:08:01.846584Z","iopub.execute_input":"2022-07-13T18:08:01.847932Z","iopub.status.idle":"2022-07-13T18:08:02.171399Z","shell.execute_reply.started":"2022-07-13T18:08:01.847898Z","shell.execute_reply":"2022-07-13T18:08:02.170603Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"GB_model =  GradientBoostingRegressor(random_state=100,\n                                      n_estimators=1000,\n                                      loss='squared_error',\n                                      subsample = 0.34,\n                                      learning_rate = 0.04)\n\nGB_model.fit(X_train_full_trans, y_train_full)\n\npreds_GB = GB_model.predict(test_X_trans)\n\npreds_GB","metadata":{"execution":{"iopub.status.busy":"2022-07-13T18:08:02.172871Z","iopub.execute_input":"2022-07-13T18:08:02.173419Z","iopub.status.idle":"2022-07-13T18:08:06.120184Z","shell.execute_reply.started":"2022-07-13T18:08:02.173387Z","shell.execute_reply":"2022-07-13T18:08:06.119387Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n\nresult_df = pd.DataFrame({'Id': test_X.index,\n                       'SalePrice': preds_GB})\nresult_df","metadata":{"execution":{"iopub.status.busy":"2022-07-13T18:08:06.121586Z","iopub.execute_input":"2022-07-13T18:08:06.121857Z","iopub.status.idle":"2022-07-13T18:08:06.136434Z","shell.execute_reply.started":"2022-07-13T18:08:06.121828Z","shell.execute_reply":"2022-07-13T18:08:06.135252Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"result_df.to_csv('submission.csv', index=False, header=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-13T18:08:06.137604Z","iopub.execute_input":"2022-07-13T18:08:06.137830Z","iopub.status.idle":"2022-07-13T18:08:06.155252Z","shell.execute_reply.started":"2022-07-13T18:08:06.137806Z","shell.execute_reply":"2022-07-13T18:08:06.154184Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# All done, please let me know if there're and suggestions!","metadata":{}}]}