{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nfrom matplotlib import pyplot as plt\n\n\nimport xgboost as xgb\nimport tensorflow as tf\nfrom tensorflow import keras\n\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import mean_squared_error, mean_absolute_error\nfrom sklearn.ensemble import RandomForestRegressor\nfrom sklearn.preprocessing import StandardScaler, Normalizer\n\n\ntry:\n    import seaborn as sns\nexcept ImportError:\n    from pip._internal import main as pip\n    pip(['install', '--user', 'seaborn'])\n    import seaborn as sns\n    \n    \npd.set_option('display.max_rows', 1000)\npd.set_option('display.max_columns', 1000)","metadata":{"id":"QgypYIzeV659","execution":{"iopub.status.busy":"2022-07-26T20:35:36.441072Z","iopub.execute_input":"2022-07-26T20:35:36.441521Z","iopub.status.idle":"2022-07-26T20:35:36.449932Z","shell.execute_reply.started":"2022-07-26T20:35:36.441435Z","shell.execute_reply":"2022-07-26T20:35:36.448691Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"np.random.seed(33)","metadata":{"id":"kGsflKzFb_fM","execution":{"iopub.status.busy":"2022-07-26T20:35:36.451719Z","iopub.execute_input":"2022-07-26T20:35:36.452310Z","iopub.status.idle":"2022-07-26T20:35:36.461962Z","shell.execute_reply.started":"2022-07-26T20:35:36.452273Z","shell.execute_reply":"2022-07-26T20:35:36.460852Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data = r'../input/house-prices-advanced-regression-techniques/train.csv'\ntest_data = r'../input/house-prices-advanced-regression-techniques/test.csv'\nsaving_folder = r''","metadata":{"id":"8QdkVCUbWTl1","execution":{"iopub.status.busy":"2022-07-26T20:35:36.464143Z","iopub.execute_input":"2022-07-26T20:35:36.464914Z","iopub.status.idle":"2022-07-26T20:35:36.474044Z","shell.execute_reply.started":"2022-07-26T20:35:36.464867Z","shell.execute_reply":"2022-07-26T20:35:36.472552Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv(data)\ntest_df = pd.read_csv(test_data)","metadata":{"id":"FV-7lffxWT-X","execution":{"iopub.status.busy":"2022-07-26T20:35:36.475993Z","iopub.execute_input":"2022-07-26T20:35:36.476368Z","iopub.status.idle":"2022-07-26T20:35:36.538775Z","shell.execute_reply.started":"2022-07-26T20:35:36.476335Z","shell.execute_reply":"2022-07-26T20:35:36.537711Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Taking a look at the data","metadata":{}},{"cell_type":"code","source":"df.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-26T20:35:36.541166Z","iopub.execute_input":"2022-07-26T20:35:36.541494Z","iopub.status.idle":"2022-07-26T20:35:36.565729Z","shell.execute_reply.started":"2022-07-26T20:35:36.541464Z","shell.execute_reply":"2022-07-26T20:35:36.564588Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Simplicity is beautiful 👌","metadata":{}},{"cell_type":"code","source":"features = df.columns\nfeatures","metadata":{"id":"AEmulAgRYMnF","outputId":"33e9ebef-2c0a-4a0f-9b43-23821acc82fa","execution":{"iopub.status.busy":"2022-07-26T20:35:36.567175Z","iopub.execute_input":"2022-07-26T20:35:36.567499Z","iopub.status.idle":"2022-07-26T20:35:36.575413Z","shell.execute_reply.started":"2022-07-26T20:35:36.567469Z","shell.execute_reply":"2022-07-26T20:35:36.574083Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Separating numeric from qualitative features","metadata":{}},{"cell_type":"code","source":"numeric = []\nqualitative_ft = []\n\nfor feature in list(features):\n  try:\n    int(df.iloc[0][feature])      #if it can't be converted to int, it's a quality\n    numeric.append(feature)\n  except ValueError:\n    qualitative_ft.append(feature)\n\nnumeric ","metadata":{"id":"rf2zC6lnrF5b","outputId":"c39f3aea-cac8-463d-9411-e1bb6512c15a","execution":{"iopub.status.busy":"2022-07-26T20:35:36.577609Z","iopub.execute_input":"2022-07-26T20:35:36.579081Z","iopub.status.idle":"2022-07-26T20:35:36.606054Z","shell.execute_reply.started":"2022-07-26T20:35:36.579027Z","shell.execute_reply":"2022-07-26T20:35:36.604556Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Removing outliers based on the price per meter square. #\nAny entry with more than 3 standard deviations should go","metadata":{"id":"Jaow4feSmzIW"}},{"cell_type":"code","source":"df[\"MeterCost\"] = df.apply(lambda row: round(row.SalePrice / row.LotArea, 2), axis=1)","metadata":{"id":"HB2vM8DcmmIm","execution":{"iopub.status.busy":"2022-07-26T20:35:36.607531Z","iopub.execute_input":"2022-07-26T20:35:36.607994Z","iopub.status.idle":"2022-07-26T20:35:36.653727Z","shell.execute_reply.started":"2022-07-26T20:35:36.607946Z","shell.execute_reply":"2022-07-26T20:35:36.651541Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mean_cost = df.MeterCost.mean()\nstd_cost = df.MeterCost.std()","metadata":{"id":"Ex-usfgnnPfs","execution":{"iopub.status.busy":"2022-07-26T20:35:36.657503Z","iopub.execute_input":"2022-07-26T20:35:36.657881Z","iopub.status.idle":"2022-07-26T20:35:36.664422Z","shell.execute_reply.started":"2022-07-26T20:35:36.657851Z","shell.execute_reply":"2022-07-26T20:35:36.663241Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_zscore(row):\n  mean = mean_cost\n  std = std_cost\n  return (row.MeterCost - mean )  / std\n\ndf[\"z_score\"] = df.apply( get_zscore, axis =1)\n\ndf_without_outliers =  df[(df.z_score >= - 3 ) & (df.z_score <= 3 ) ]\ndf_without_outliers.reset_index(inplace=True) \n\ndf_without_outliers.shape","metadata":{"id":"OBbBvCZZnIkU","outputId":"ad280233-260e-4da3-a6c6-ae6b6502642a","execution":{"iopub.status.busy":"2022-07-26T20:35:36.669549Z","iopub.execute_input":"2022-07-26T20:35:36.670664Z","iopub.status.idle":"2022-07-26T20:35:36.708672Z","shell.execute_reply.started":"2022-07-26T20:35:36.670539Z","shell.execute_reply":"2022-07-26T20:35:36.707532Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Doing something about the missing values 🔎\n","metadata":{}},{"cell_type":"markdown","source":"Here's my strategy:\n\nI'm **dropping**: Fence, Alley, PoolQC  because they have too few entries, and also id because it doesn't interest us\n\nI'm filling the other columns like this:\n\n**median** = numeric features\n\n**constant** = MiscFeature \n\n**most_common** = qualitative features","metadata":{"id":"oBYf-owO_gCi"}},{"cell_type":"code","source":"def df_fill_missing_values(dataframe: pd.DataFrame, strategy_columns: dict, fill_value='unknown'):\n  \"\"\"Fills missing values in dataframes columns using Sklearn Simple Imputer. \n  Params = DataFrame, and dict of imputer_strategy(str) and column_names iterable pairs\n  strategoies: mean, median, most_frequent or constant (uses fill value)\"\"\"\n  from sklearn.impute import SimpleImputer\n  df = dataframe\n  for imputer_strategy, column_names in strategy_columns.items():\n    for column_name in column_names:\n      imputer = SimpleImputer(strategy=imputer_strategy, fill_value=fill_value)\n      incomplete = np.array(df[column_name]).reshape(-1, 1)\n      filled = imputer.fit(incomplete).transform(incomplete)\n      filled = pd.DataFrame(filled, columns=[column_name])\n      df = df.drop(column_name, axis=1).join(filled)\n  return df\n","metadata":{"id":"llgy7HdEWUPj","execution":{"iopub.status.busy":"2022-07-26T20:35:36.710081Z","iopub.execute_input":"2022-07-26T20:35:36.710400Z","iopub.status.idle":"2022-07-26T20:35:36.717516Z","shell.execute_reply.started":"2022-07-26T20:35:36.710372Z","shell.execute_reply":"2022-07-26T20:35:36.716586Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def standardize_features( dataframe: pd.DataFrame, feature_names: list, operation=\"standardize\"):\n  \"\"\"standardizes numerical features and returns a dataframe\"\"\"\n  from sklearn.preprocessing import StandardScaler, Normalizer\n  if operation == \"normalize\":\n    operator = Normalizer()\n  else:\n    operator = StandardScaler()\n  df = dataframe\n  for column_name in feature_names:\n      raw_data = np.array(df[column_name]).reshape(-1, 1)\n      transformed = operator.fit_transform(raw_data)\n      ready = pd.DataFrame(transformed, columns=[column_name])\n      df = df.drop(column_name, axis=1).join(ready)\n  return df\n\n#  ENDED UP NOT EVEN USING THIS BECAUSE IT DOESN'T AFFECT RANDOM FORESTS OR XGB, AND A LAYER OF THE NEURAL NETWORK DOES THIS","metadata":{"id":"CXyGnCOjZrpr","execution":{"iopub.status.busy":"2022-07-26T20:35:36.718865Z","iopub.execute_input":"2022-07-26T20:35:36.719441Z","iopub.status.idle":"2022-07-26T20:35:36.730524Z","shell.execute_reply.started":"2022-07-26T20:35:36.719408Z","shell.execute_reply":"2022-07-26T20:35:36.729068Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Notice that the testing data also has missing values, so it needs its own strategy similar to the training data's","metadata":{}},{"cell_type":"code","source":"TO_DELETE = ['Alley', 'PoolQC', 'Fence', 'Id']\n\nclass Preprocess:\n  \n  incomplete_train_fts =  [feat for feat in list(df.columns) if df[feat].isnull().values.any()]\n\n  strategy1 = {\"most_frequent\": [ft for ft in incomplete_train_fts if ft in qualitative_ft and ft not in TO_DELETE]\n              ,\"median\": [ft for ft in incomplete_train_fts if ft in numeric],\n              \"constant\": [\"MiscFeature\"]\n              }\n\n  incomplete_test_fts = [feat for feat in list(test_df.columns) if test_df[feat].isnull().values.any()]\n  \n  strategy2 = {\"most_frequent\": [ft for ft in incomplete_test_fts if ft in qualitative_ft and ft not in [\"MiscFeature\",\"LotFrontage\", \"MasVnrArea\",\"Alley\", \"Fence\", \"PoolQC\", \"Id\" ]],\n               \"median\": [ft for ft in incomplete_test_fts if ft in numeric],\n               \"constant\": [\"MiscFeature\"]\n  }\n\n\n  def fit_transform(self, dataframe, test_data=False):\n    to_delete = TO_DELETE\n    if test_data:\n      strategy = self.strategy2\n    else:\n      to_delete = TO_DELETE + [\"MeterCost\", \"z_score\"]\n      strategy = self.strategy1\n    df = dataframe.drop(to_delete, axis=1)\n    filled = df_fill_missing_values(df, strategy)\n    return filled\n","metadata":{"id":"eU-0SZvaWUSp","execution":{"iopub.status.busy":"2022-07-26T20:35:36.732387Z","iopub.execute_input":"2022-07-26T20:35:36.732785Z","iopub.status.idle":"2022-07-26T20:35:36.790662Z","shell.execute_reply.started":"2022-07-26T20:35:36.732754Z","shell.execute_reply":"2022-07-26T20:35:36.789291Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preprocess = Preprocess()\nfilled = preprocess.fit_transform(df).drop(\"SalePrice\", axis=1)   # change dataframe if want to remove the outliers\nfilled_test = preprocess.fit_transform(test_df, test_data=True)","metadata":{"id":"EdQ_rw66WUWJ","execution":{"iopub.status.busy":"2022-07-26T20:35:36.792393Z","iopub.execute_input":"2022-07-26T20:35:36.792990Z","iopub.status.idle":"2022-07-26T20:35:37.160083Z","shell.execute_reply.started":"2022-07-26T20:35:36.792936Z","shell.execute_reply":"2022-07-26T20:35:37.159056Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(filled.shape,\nfilled_test.shape)","metadata":{"execution":{"iopub.status.busy":"2022-07-26T20:35:37.162070Z","iopub.execute_input":"2022-07-26T20:35:37.162544Z","iopub.status.idle":"2022-07-26T20:35:37.168747Z","shell.execute_reply.started":"2022-07-26T20:35:37.162500Z","shell.execute_reply":"2022-07-26T20:35:37.167513Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Maybe we can discover something looking at correlation🤔","metadata":{"id":"4K1gHrHVMUPM"}},{"cell_type":"code","source":"fig, ax = plt.subplots(figsize=(11, 10))\nax = sns.heatmap(filled.corr(), vmin=-1, vmax=1, \n                 cmap=sns.diverging_palette(20, 220, as_cmap=True),\n                 ax=ax)\n\nplt.tight_layout()\nplt.show()","metadata":{"id":"_gLAO8SrHyqS","outputId":"5b3779ce-ba7f-4427-8435-51ce16847299","execution":{"iopub.status.busy":"2022-07-26T20:35:37.170219Z","iopub.execute_input":"2022-07-26T20:35:37.171518Z","iopub.status.idle":"2022-07-26T20:35:38.089787Z","shell.execute_reply.started":"2022-07-26T20:35:37.171475Z","shell.execute_reply":"2022-07-26T20:35:38.088543Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Way too many features! looking at this hurts my brain. \nAfter we encode the qualitative variables we should be able to use another\ntechnique to see the most important variables","metadata":{}},{"cell_type":"markdown","source":"# Making X and Y for our model #\n\n# One hot encoding\n\nI concatenated the training data and the test data to use one hot encoding on the qualitative features, then separated them again\n\n","metadata":{}},{"cell_type":"code","source":"for ft in TO_DELETE:\n  try:\n    qualitative_ft.remove(ft)\n  except ValueError:\n    pass\n\ntemp_joined = pd.concat([filled, filled_test], ignore_index=True, axis=0)\n\ntry:\n  temp_joined = temp_joined.drop([\"index\"], axis=1)\nexcept KeyError:\n  pass\n\n\nencoded = pd.get_dummies(temp_joined, prefix='', prefix_sep='', columns=qualitative_ft, drop_first=True)\n\nx = encoded.iloc[0:1460]                           # ADJUST WHEN NEEDED - if outliers are removed : df_without_outliers.shape\ny = df.SalePrice                                   # ADJUST WHEN NEEDED - df_without_outliers\n\ntesting_data = encoded.iloc[1460:]                  # ADJUST WHEN NEEDED - df_without_outliers","metadata":{"id":"wUowcNSNQNJm","execution":{"iopub.status.busy":"2022-07-26T20:35:38.344751Z","iopub.execute_input":"2022-07-26T20:35:38.346588Z","iopub.status.idle":"2022-07-26T20:35:38.477643Z","shell.execute_reply.started":"2022-07-26T20:35:38.346535Z","shell.execute_reply":"2022-07-26T20:35:38.476165Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_train_full, x_test, y_train_full, y_test =  train_test_split(x, y, test_size=0.1)                  \nx_valid, x_train =  x_train_full[:135], x_train_full[135:]\ny_valid, y_train =  y_train_full[:135], y_train_full[135:]","metadata":{"id":"2sGprEIeYc0y","execution":{"iopub.status.busy":"2022-07-26T20:35:38.479227Z","iopub.execute_input":"2022-07-26T20:35:38.479551Z","iopub.status.idle":"2022-07-26T20:35:38.491404Z","shell.execute_reply.started":"2022-07-26T20:35:38.479523Z","shell.execute_reply":"2022-07-26T20:35:38.490283Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# After fixing the data, let's use Principal Component Analysis to see the most important features","metadata":{}},{"cell_type":"code","source":"from sklearn.decomposition import PCA\n\n\ndef find_n_components(X_train, preserved_variance):\n    pca = PCA()\n    pca.fit(X_train)\n    cumsum = np.cumsum(pca.explained_variance_ratio_)\n    return np.argmax(cumsum >= preserved_variance) + 1\n\n\ndef drop_dimensions(x_train, preserved_variance):\n    \"\"\"retain n% of variance , while ignoring the rest of the data\"\"\"\n    pca = PCA(n_components=preserved_variance,)\n    return pca.fit_transform(x_train)\n\n\ndef pca_frame(x_train, components=10):\n    pca = PCA(n_components=components)\n    pca.fit_transform(x_train)\n    indexes = [f\"PCA-{n}\" for n in range(1, components + 1)]\n    return pd.DataFrame(pca.components_, columns=x_train.columns, index=indexes)\n","metadata":{"id":"tB5qiX7XQ3DL","execution":{"iopub.status.busy":"2022-07-26T20:35:38.091103Z","iopub.execute_input":"2022-07-26T20:35:38.091961Z","iopub.status.idle":"2022-07-26T20:35:38.101134Z","shell.execute_reply.started":"2022-07-26T20:35:38.091925Z","shell.execute_reply":"2022-07-26T20:35:38.099186Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"find_n_components(x, preserved_variance= 0.9999)     \n\n","metadata":{"id":"-HovQ4VLQaEv","outputId":"a61677aa-3dae-4272-9f89-276b32bdc749","execution":{"iopub.status.busy":"2022-07-26T20:35:38.102916Z","iopub.execute_input":"2022-07-26T20:35:38.103454Z","iopub.status.idle":"2022-07-26T20:35:38.176173Z","shell.execute_reply.started":"2022-07-26T20:35:38.103391Z","shell.execute_reply":"2022-07-26T20:35:38.174927Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"PCA says  very few features are actually important and we still preserve 99.99% of the information. ","metadata":{}},{"cell_type":"code","source":"reduced = drop_dimensions(x_train, preserved_variance= 0.9999)\n\n# JUST IN CASE","metadata":{"id":"XMfsGYrpQaHf","execution":{"iopub.status.busy":"2022-07-26T20:35:38.178261Z","iopub.execute_input":"2022-07-26T20:35:38.179187Z","iopub.status.idle":"2022-07-26T20:35:38.239997Z","shell.execute_reply.started":"2022-07-26T20:35:38.179131Z","shell.execute_reply":"2022-07-26T20:35:38.238720Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# The top 20 features 🏆\n(according to PCA)","metadata":{}},{"cell_type":"code","source":"pca_results = pca_frame(x_train, components=20)\nbest_features = list(pca_results.idxmax(axis=1))\nbest_features","metadata":{"id":"wXD6N8iRQaSh","outputId":"1b750df5-e5a9-4268-c529-bc0c4eeb706a","execution":{"iopub.status.busy":"2022-07-26T20:35:38.241965Z","iopub.execute_input":"2022-07-26T20:35:38.242803Z","iopub.status.idle":"2022-07-26T20:35:38.341752Z","shell.execute_reply.started":"2022-07-26T20:35:38.242751Z","shell.execute_reply":"2022-07-26T20:35:38.340150Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Why I used the full data instead\nI ended up using the full data because with the reduced data the results were poor.\nI guess my outlier removing strategy was bad or unnecessary and reducing the data's dimension removes some info that gets picked up by \nthe models","metadata":{}},{"cell_type":"markdown","source":"# XGBoost and Random Forests","metadata":{"id":"yMIw_1PpdYA8"}},{"cell_type":"code","source":"forests = RandomForestRegressor(random_state=42, n_estimators=200)\nxgbreg = xgb.XGBRegressor(n_estimators = 300, objective='reg:squarederror', learning_rate=0.10, random_state=33, max_depth=8 ) \n\n\nselected_model = xgbreg\n\ntrain_array = x_train_full.to_numpy()\ntest_array = x_test.to_numpy()\n\n\nselected_model.fit(train_array, y_train_full)\npred = selected_model.predict(test_array)\nsquared_error = mean_squared_error(pred, y_test, squared=False)\nabsolute_error = mean_absolute_error(pred, y_test)\nprint(\"XGB errors: \", absolute_error, squared_error)","metadata":{"id":"HA5xmaZ9Yc31","outputId":"90ecf84c-d667-4b61-807a-1a8bedd7e30d","execution":{"iopub.status.busy":"2022-07-26T21:00:14.854817Z","iopub.execute_input":"2022-07-26T21:00:14.855195Z","iopub.status.idle":"2022-07-26T21:00:18.843272Z","shell.execute_reply.started":"2022-07-26T21:00:14.855165Z","shell.execute_reply":"2022-07-26T21:00:18.842128Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"forests.fit(train_array, y_train_full)\npred = forests.predict(test_array)\nsquared_error = mean_squared_error(pred, y_test, squared=False)\nabsolute_error = mean_absolute_error(pred, y_test)\nprint(\"Forests: \", absolute_error, squared_error)","metadata":{"execution":{"iopub.status.busy":"2022-07-26T20:59:55.497342Z","iopub.execute_input":"2022-07-26T20:59:55.497725Z","iopub.status.idle":"2022-07-26T21:00:00.717090Z","shell.execute_reply.started":"2022-07-26T20:59:55.497690Z","shell.execute_reply":"2022-07-26T21:00:00.715863Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Automatic hyperparameter search with GridSearch","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import RandomizedSearchCV\n\n\nxgb_param_distribs = {\n\"n_estimators\" : list(range(150, 300)),\n\"learning_rate\": np.arange(start=0.03, stop=0.5, step=0.005),\n\"max_depth\": list(range(1,9)),\n}\n\nsearcher = RandomizedSearchCV(xgbreg, xgb_param_distribs, n_iter=10, cv=5)\nsearcher.fit(train_array, y_train_full)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-26T20:35:42.589729Z","iopub.execute_input":"2022-07-26T20:35:42.590039Z","iopub.status.idle":"2022-07-26T20:37:16.113076Z","shell.execute_reply.started":"2022-07-26T20:35:42.590010Z","shell.execute_reply":"2022-07-26T20:37:16.111923Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"searcher.best_params_","metadata":{"execution":{"iopub.status.busy":"2022-07-26T20:37:16.117352Z","iopub.execute_input":"2022-07-26T20:37:16.117712Z","iopub.status.idle":"2022-07-26T20:37:16.125100Z","shell.execute_reply.started":"2022-07-26T20:37:16.117681Z","shell.execute_reply":"2022-07-26T20:37:16.124018Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"best_xgb = searcher.best_estimator_\nbest_xgb","metadata":{"execution":{"iopub.status.busy":"2022-07-26T21:01:06.093072Z","iopub.execute_input":"2022-07-26T21:01:06.093990Z","iopub.status.idle":"2022-07-26T21:01:06.104280Z","shell.execute_reply.started":"2022-07-26T21:01:06.093951Z","shell.execute_reply":"2022-07-26T21:01:06.103060Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"best_xgb.fit(train_array, y_train_full)\npred = best_xgb.predict(test_array)\nsquared_error = mean_squared_error(pred, y_test, squared=False)\nabsolute_error = mean_absolute_error(pred, y_test)\nprint(\"Best XGB error: \", absolute_error, squared_error)","metadata":{"execution":{"iopub.status.busy":"2022-07-26T21:01:08.630738Z","iopub.execute_input":"2022-07-26T21:01:08.631103Z","iopub.status.idle":"2022-07-26T21:01:10.887154Z","shell.execute_reply.started":"2022-07-26T21:01:08.631074Z","shell.execute_reply":"2022-07-26T21:01:10.885983Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"~17k absolute error. The same model gives a score of  ~ 0.13745 on the submission data","metadata":{}},{"cell_type":"code","source":"forest_param_distribs = {\n\"n_estimators\" : list(range(150, 300)),\n\"min_samples_leaf\": [1,2,3],\n\"max_depth\": list(range(0,10)),\n}\n\nforest_searcher = RandomizedSearchCV(forests, forest_param_distribs, n_iter=10, cv=5)\nforest_searcher.fit(train_array, y_train_full)","metadata":{"execution":{"iopub.status.busy":"2022-07-26T20:52:09.777087Z","iopub.execute_input":"2022-07-26T20:52:09.777543Z","iopub.status.idle":"2022-07-26T20:54:08.303778Z","shell.execute_reply.started":"2022-07-26T20:52:09.777506Z","shell.execute_reply":"2022-07-26T20:54:08.302684Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"best_forest = forest_searcher.best_estimator_\nbest_forest.fit(train_array, y_train_full)\npred = best_forest.predict(test_array)\nsquared_error = mean_squared_error(pred, y_test, squared=False)\nabsolute_error = mean_absolute_error(pred, y_test)\nprint(\"forest error\", absolute_error, squared_error)","metadata":{"execution":{"iopub.status.busy":"2022-07-26T21:01:37.414739Z","iopub.execute_input":"2022-07-26T21:01:37.415100Z","iopub.status.idle":"2022-07-26T21:01:40.596888Z","shell.execute_reply.started":"2022-07-26T21:01:37.415071Z","shell.execute_reply":"2022-07-26T21:01:40.594765Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#  Would a neural will network perform better?  ","metadata":{"id":"wlELzRWSddd-"}},{"cell_type":"code","source":"\ntf.random.set_seed(42)\n\nearly_stopping_cb = keras.callbacks.EarlyStopping(patience=20, restore_best_weights=True, monitor=\"loss\")\n\nepochs = 200\n\n\nsample_to_normalize = x_train\nnorm_layer = keras.layers.Normalization(axis=1)\nnorm_layer.adapt(tf.constant(sample_to_normalize))\n\nneural = keras.models.Sequential([\nnorm_layer,\nkeras.layers.Dense(240, activation=\"selu\", kernel_initializer=\"lecun_normal\", input_shape=x_train.shape[1:]),\nkeras.layers.Dropout(rate=0.6),\nkeras.layers.Dense(240, kernel_initializer=\"lecun_normal\", activation=\"selu\"),\nkeras.layers.Dense(1, activation=None),\n])\n\noptimizer = keras.optimizers.SGD(learning_rate=0.0005, nesterov=True) #keras.optimizers.RMSprop(lr=0.001) \n#optimizer = keras.optimizers.Nadam(learning_rate=0.0005)\nloss = tf.keras.losses.MeanAbsoluteError()\nneural.compile(loss=loss, optimizer= optimizer)\n\nhistory = neural.fit(x_train, y_train, validation_data=(x_valid, y_valid), epochs=epochs, callbacks=[early_stopping_cb])","metadata":{"id":"O1vY4AvHYc6o","outputId":"736de1b1-f102-4a2e-d4f8-68612d7692a2","execution":{"iopub.status.busy":"2022-07-26T20:37:18.199177Z","iopub.execute_input":"2022-07-26T20:37:18.200038Z","iopub.status.idle":"2022-07-26T20:37:18.518329Z","shell.execute_reply.started":"2022-07-26T20:37:18.199991Z","shell.execute_reply":"2022-07-26T20:37:18.517390Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nloss = history.history['loss']\nval_loss = history.history['val_loss']\nepochs_range = range(len(history.epoch))\nplt.figure(figsize=(8, 8))\nplt.plot(epochs_range, loss, label='Training Loss')\nplt.plot(epochs_range, val_loss, label='Validation Loss')\nplt.legend(loc='upper right')\nplt.title('Training and Validation Loss')\nplt.show()","metadata":{"id":"EpmezPKA9tQq","outputId":"e56dda19-b11e-4bf8-fa1b-d32fc4461a2e","execution":{"iopub.status.busy":"2022-07-26T20:37:18.519601Z","iopub.execute_input":"2022-07-26T20:37:18.519949Z","iopub.status.idle":"2022-07-26T20:37:18.741922Z","shell.execute_reply.started":"2022-07-26T20:37:18.519919Z","shell.execute_reply":"2022-07-26T20:37:18.740704Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"As you can see, this neural network sucks 🤔 its error is higher than that of XGboost and RandomForests.","metadata":{}},{"cell_type":"markdown","source":"# Conclusion \n\nOur XGB model works best in this case, and RandomForest was close. This dataset is overcrowded with features but the entries are too few for the neural network I came up with. \n\nI think my outlier removal strategy wasn't the best so the score was better without it. Using only the top 20 features performed worse in my tests as well.\n","metadata":{}},{"cell_type":"markdown","source":"# Making a predictions file to the kaggle","metadata":{"id":"hvKEKTc8uL0T"}},{"cell_type":"code","source":"from IPython.display import HTML\n\ndef kaggle_submission(trained_model, test_data, savename=\"my_predictions\", neural=False):\n    predictions = trained_model.predict(test_data)\n    if neural:\n        predictions = [pred for [pred] in predictions]\n    indices = list(range(1461, 2920))\n    sub_df = pd.DataFrame(zip(indices, predictions), columns=[\"Id\",\"SalePrice\"])\n    name = f\"{savename}.csv\"\n    sub_df.to_csv(name, index=False)\n    html = '<a href={filename}>{title}</a>'\n    html = html.format(title=\"Download CSV file\",filename=name)\n    return HTML(html)","metadata":{"id":"P1KYAzZTYdAT","execution":{"iopub.status.busy":"2022-07-26T20:37:18.743785Z","iopub.execute_input":"2022-07-26T20:37:18.744328Z","iopub.status.idle":"2022-07-26T20:37:18.753407Z","shell.execute_reply.started":"2022-07-26T20:37:18.744286Z","shell.execute_reply":"2022-07-26T20:37:18.752105Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"kaggle_submission(best_xgb, testing_data.to_numpy(), \"xgbreg\", neural=False)","metadata":{"id":"KU4Wu-1KL8x6","execution":{"iopub.status.busy":"2022-07-26T20:37:18.754936Z","iopub.execute_input":"2022-07-26T20:37:18.755297Z","iopub.status.idle":"2022-07-26T20:37:18.792888Z","shell.execute_reply.started":"2022-07-26T20:37:18.755266Z","shell.execute_reply":"2022-07-26T20:37:18.791961Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"id":"HsedGLy2hAsf","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"id":"KWZDNPjbhAvQ"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"id":"-RPbelS5hAyp"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"id":"Dc9VRNP9hA17"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"id":"n8kilRVWhA5I"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"id":"BCBh-wzihA7A"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"id":"elZkWcbUhA9w"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"id":"GH1Xz9zchBA9"},"execution_count":null,"outputs":[]}]}