{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# import needed libraries\n\nimport pandas as pd\nimport numpy as np\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.feature_selection import mutual_info_regression\nfrom scipy import stats\nfrom sklearn.compose import ColumnTransformer\nfrom sklearn.preprocessing import OneHotEncoder\nimport math\nfrom sklearn.ensemble import RandomForestRegressor\nfrom scipy.stats import pearsonr","metadata":{"execution":{"iopub.status.busy":"2022-07-12T13:45:15.229236Z","iopub.execute_input":"2022-07-12T13:45:15.229691Z","iopub.status.idle":"2022-07-12T13:45:15.236981Z","shell.execute_reply.started":"2022-07-12T13:45:15.229657Z","shell.execute_reply":"2022-07-12T13:45:15.235788Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 1. Pre-processing","metadata":{}},{"cell_type":"markdown","source":"**Method used**\n- Filling missing values\n- Cardinality check\n- Feauture selection with mutual regression\n- Removing ouliers\n- Encoding","metadata":{}},{"cell_type":"markdown","source":"## 1.1 Data loading and preview","metadata":{}},{"cell_type":"code","source":"test_data = pd.read_csv(\"/kaggle/input/house-prices-advanced-regression-techniques/test.csv\")\ntrain_data = pd.read_csv(\"/kaggle/input/house-prices-advanced-regression-techniques/train.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-07-12T13:45:15.238936Z","iopub.execute_input":"2022-07-12T13:45:15.239774Z","iopub.status.idle":"2022-07-12T13:45:15.294483Z","shell.execute_reply.started":"2022-07-12T13:45:15.239736Z","shell.execute_reply":"2022-07-12T13:45:15.293420Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train data overview\n\ntrain_data.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-12T13:45:15.297327Z","iopub.execute_input":"2022-07-12T13:45:15.297739Z","iopub.status.idle":"2022-07-12T13:45:15.325228Z","shell.execute_reply.started":"2022-07-12T13:45:15.297702Z","shell.execute_reply":"2022-07-12T13:45:15.324355Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-12T13:45:15.326533Z","iopub.execute_input":"2022-07-12T13:45:15.327047Z","iopub.status.idle":"2022-07-12T13:45:15.333041Z","shell.execute_reply.started":"2022-07-12T13:45:15.327015Z","shell.execute_reply":"2022-07-12T13:45:15.332011Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-12T13:45:15.335470Z","iopub.execute_input":"2022-07-12T13:45:15.336059Z","iopub.status.idle":"2022-07-12T13:45:15.367795Z","shell.execute_reply.started":"2022-07-12T13:45:15.336026Z","shell.execute_reply":"2022-07-12T13:45:15.366583Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-12T13:45:15.369203Z","iopub.execute_input":"2022-07-12T13:45:15.370064Z","iopub.status.idle":"2022-07-12T13:45:15.377509Z","shell.execute_reply.started":"2022-07-12T13:45:15.370027Z","shell.execute_reply":"2022-07-12T13:45:15.376137Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 1.2 Handle Missing values","metadata":{}},{"cell_type":"code","source":"# filling missing value (Numerical and Categorical)\n\nfor col in train_data.columns:\n    if train_data[col].dtype in [np.float64, np.int64]:\n        sip = SimpleImputer(missing_values=np.nan, strategy='mean')\n        train_data[col] = sip.fit_transform(train_data[col].values.reshape(-1, 1))[:, 0]\n    else:\n        sip = SimpleImputer(missing_values=np.nan, strategy='constant', fill_value=\"Not_specified\")\n        train_data[col] = sip.fit_transform(train_data[col].values.reshape(-1, 1))[:, 0]\n\nfor col in test_data.columns:\n    if test_data[col].dtype in [np.float64, np.int64]:\n        sip = SimpleImputer(missing_values=np.nan, strategy='mean')\n        test_data[col] = sip.fit_transform(test_data[col].values.reshape(-1, 1))[:, 0]\n    else:\n        sip = SimpleImputer(missing_values=np.nan, strategy='constant', fill_value=\"Not_specified\")\n        test_data[col] = sip.fit_transform(test_data[col].values.reshape(-1, 1))[:, 0]","metadata":{"execution":{"iopub.status.busy":"2022-07-12T13:45:15.379097Z","iopub.execute_input":"2022-07-12T13:45:15.379535Z","iopub.status.idle":"2022-07-12T13:45:15.508329Z","shell.execute_reply.started":"2022-07-12T13:45:15.379489Z","shell.execute_reply":"2022-07-12T13:45:15.506287Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 1.3 Feature and ouput seperation","metadata":{}},{"cell_type":"code","source":"# seperate features and outcome (SalePrice)\n\nX_train = train_data.iloc[:, :-1]\ny_train = train_data.iloc[:, -1]\nX_test = test_data","metadata":{"execution":{"iopub.status.busy":"2022-07-12T13:45:15.514507Z","iopub.execute_input":"2022-07-12T13:45:15.515508Z","iopub.status.idle":"2022-07-12T13:45:15.529981Z","shell.execute_reply.started":"2022-07-12T13:45:15.515445Z","shell.execute_reply":"2022-07-12T13:45:15.528465Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 1.4 Cardinality and Dimension reduction","metadata":{}},{"cell_type":"code","source":"# get each column cardinality (except the last column \"SalePrice\")\n\n# we romve Id column first\nX_train.drop(['Id'], axis=1, inplace=True)\nX_test.drop(['Id'], axis=1, inplace=True)\n\nhigh_cardinality_columns = []\ncardinality_result = pd.DataFrame(columns=X_test.columns)\nrow = {}\n# cardinality divided by total number of records\ncardinality_threshold = 100\nfor col in X_train.columns:\n    card = len(set(X_train[col]))\n    row[col] = card\n    if card > cardinality_threshold and X_train[col].dtype not in [np.float64, np.int64]:\n        high_cardinality_columns.append(col)\n\ncardinality_result = cardinality_result.append(row, ignore_index=True)\ncardinality_result.drop(high_cardinality_columns, axis=1, inplace=True)\ncardinality_result","metadata":{"execution":{"iopub.status.busy":"2022-07-12T13:45:15.531754Z","iopub.execute_input":"2022-07-12T13:45:15.532143Z","iopub.status.idle":"2022-07-12T13:45:15.601440Z","shell.execute_reply.started":"2022-07-12T13:45:15.532110Z","shell.execute_reply":"2022-07-12T13:45:15.600628Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(high_cardinality_columns)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T13:45:15.603362Z","iopub.execute_input":"2022-07-12T13:45:15.603945Z","iopub.status.idle":"2022-07-12T13:45:15.612547Z","shell.execute_reply.started":"2022-07-12T13:45:15.603899Z","shell.execute_reply":"2022-07-12T13:45:15.611315Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-12T13:45:15.617216Z","iopub.execute_input":"2022-07-12T13:45:15.618148Z","iopub.status.idle":"2022-07-12T13:45:15.655488Z","shell.execute_reply.started":"2022-07-12T13:45:15.618090Z","shell.execute_reply":"2022-07-12T13:45:15.654543Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 1.5 Feature selection with Mutual Information","metadata":{}},{"cell_type":"code","source":"# we calculate the mutual information score between features and outcome (salePrice)\n\noutput = y_train.values.reshape(-1, 1)\nfeature_to_drop = []\nmi_res = {}\ndropping_threshold = 0.05\nfor col in X_train.columns:\n    if X_train[col].dtype in [np.float64, np.int64]:\n        mi_res[col] = mutual_info_regression(output, X_train[col], random_state=0)[0]\n    else:\n        f_factorized,_ = X_train[col].factorize()\n        mi_res[col] = mutual_info_regression(output, f_factorized, random_state=0)[0]\n        \n    if mi_res[col] < dropping_threshold:\n            feature_to_drop.append(col)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T13:45:15.657280Z","iopub.execute_input":"2022-07-12T13:45:15.657645Z","iopub.status.idle":"2022-07-12T13:45:16.521367Z","shell.execute_reply.started":"2022-07-12T13:45:15.657609Z","shell.execute_reply":"2022-07-12T13:45:16.520288Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train.drop(feature_to_drop, axis=1, inplace=True)\nX_test.drop(feature_to_drop, axis=1, inplace=True)\nX_train.head()","metadata":{"scrolled":true,"execution":{"iopub.status.busy":"2022-07-12T13:45:16.523426Z","iopub.execute_input":"2022-07-12T13:45:16.524090Z","iopub.status.idle":"2022-07-12T13:45:16.562547Z","shell.execute_reply.started":"2022-07-12T13:45:16.524033Z","shell.execute_reply":"2022-07-12T13:45:16.561364Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-12T13:45:16.563953Z","iopub.execute_input":"2022-07-12T13:45:16.564353Z","iopub.status.idle":"2022-07-12T13:45:16.570596Z","shell.execute_reply.started":"2022-07-12T13:45:16.564321Z","shell.execute_reply":"2022-07-12T13:45:16.569682Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 1.6 Feature Selection with Pearson Correlation","metadata":{}},{"cell_type":"code","source":"# pearson correlation feature select\n\ny = list(y_train)\np_value_threhold = 0.05\nmin_corr_score = 0.25\ncolumn_to_drop = []\nfor col in X_train.columns:\n    if X_train[col].dtype not in [np.float64, np.int64]:\n        col_values,_ = X_train[col].factorize()\n    else:\n        col_values = list(X_train[col].values)\n    corr = pearsonr(col_values, y)\n    if corr[1] > p_value_threhold or abs(corr[0]) < min_corr_score:\n        column_to_drop.append(col)\n\ncolumn_to_drop","metadata":{"execution":{"iopub.status.busy":"2022-07-12T13:45:16.571628Z","iopub.execute_input":"2022-07-12T13:45:16.571949Z","iopub.status.idle":"2022-07-12T13:45:16.623523Z","shell.execute_reply.started":"2022-07-12T13:45:16.571921Z","shell.execute_reply":"2022-07-12T13:45:16.622663Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train.drop(column_to_drop, axis=1, inplace=True)\nX_test.drop(column_to_drop, axis=1, inplace=True)\nX_train.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-12T13:45:16.624797Z","iopub.execute_input":"2022-07-12T13:45:16.625102Z","iopub.status.idle":"2022-07-12T13:45:16.661993Z","shell.execute_reply.started":"2022-07-12T13:45:16.625075Z","shell.execute_reply":"2022-07-12T13:45:16.661055Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-12T13:45:16.662874Z","iopub.execute_input":"2022-07-12T13:45:16.663162Z","iopub.status.idle":"2022-07-12T13:45:16.670796Z","shell.execute_reply.started":"2022-07-12T13:45:16.663135Z","shell.execute_reply":"2022-07-12T13:45:16.669810Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 1.7 Outlier detection","metadata":{}},{"cell_type":"code","source":"rows_to_drop = []\nfor col in X_train.columns:\n    if X_train[col].dtype not in [np.float64, np.int64]:\n        continue\n    z_scores = stats.zscore(X_train[col])\n    for i in range(len(z_scores)):\n        if abs(z_scores[i]) > 3:\n            rows_to_drop.append(i)\n\nrows_to_drop = list(set(rows_to_drop))\nX_train = X_train.drop(rows_to_drop)\ny_train = y_train.drop(rows_to_drop)\nX_train.reset_index(drop=True, inplace=True)\ny_train.reset_index(drop=True, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T13:45:16.671852Z","iopub.execute_input":"2022-07-12T13:45:16.672224Z","iopub.status.idle":"2022-07-12T13:45:16.866908Z","shell.execute_reply.started":"2022-07-12T13:45:16.672178Z","shell.execute_reply":"2022-07-12T13:45:16.865708Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-12T13:45:16.868486Z","iopub.execute_input":"2022-07-12T13:45:16.868852Z","iopub.status.idle":"2022-07-12T13:45:16.875257Z","shell.execute_reply.started":"2022-07-12T13:45:16.868821Z","shell.execute_reply":"2022-07-12T13:45:16.874029Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 1.8 Encoding","metadata":{}},{"cell_type":"code","source":"categorical_indexes = []\nfor col in X_train.columns:\n    if X_train[col].dtype not in [np.float64, np.int64]:\n        categorical_indexes.append(list(X_train.columns).index(col))\n\nct = ColumnTransformer(transformers=[('encoder', OneHotEncoder(), categorical_indexes)], remainder='passthrough')","metadata":{"execution":{"iopub.status.busy":"2022-07-12T13:45:16.876880Z","iopub.execute_input":"2022-07-12T13:45:16.877271Z","iopub.status.idle":"2022-07-12T13:45:16.886673Z","shell.execute_reply.started":"2022-07-12T13:45:16.877236Z","shell.execute_reply":"2022-07-12T13:45:16.885669Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train = X_train.iloc[:, :].values\nX_test = X_test.iloc[:, :].values\nX_train = ct.fit_transform(X_train)\nX_test = ct.fit_transform(X_test)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T13:45:16.887679Z","iopub.execute_input":"2022-07-12T13:45:16.888018Z","iopub.status.idle":"2022-07-12T13:45:16.924870Z","shell.execute_reply.started":"2022-07-12T13:45:16.887980Z","shell.execute_reply":"2022-07-12T13:45:16.923637Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 2. Random Forest Regression","metadata":{}},{"cell_type":"code","source":"rf_regressor = RandomForestRegressor(n_estimators=4000, random_state=0)\nrf_regressor.fit(X_train, y_train)\ny_pred_rf = rf_regressor.predict(X_test)\nsubmission = pd.DataFrame(columns=['id', 'SalePrice'])\ntest_data = pd.read_csv(\"/kaggle/input/house-prices-advanced-regression-techniques/test.csv\")\nfor i in range(len(y_pred_rf)):\n    row = {}\n    row['id'] = int(test_data.iloc[i]['Id'])\n    row['SalePrice'] = y_pred_rf[i]    \n    submission = submission.append(row, ignore_index=True)\n\nsubmission['id'] = submission['id'].astype(int)\nsubmission['SalePrice'] = submission['SalePrice'].astype(float)\nsubmission.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-12T13:45:16.926479Z","iopub.execute_input":"2022-07-12T13:45:16.927221Z","iopub.status.idle":"2022-07-12T13:46:16.767347Z","shell.execute_reply.started":"2022-07-12T13:45:16.927159Z","shell.execute_reply":"2022-07-12T13:46:16.766225Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}