{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Housing Analysis With Sklearn Ensembling + Tensorflow","metadata":{}},{"cell_type":"markdown","source":"# Overview\nGeneral structure of this notebook: \n- **Missing Data** - Here we examine graphically the amount of missing data and using differing methods to deal with the missing data. \n- **Feature Engineering** - Simple binning of values for easier categorization and visualization. Not many new features were created as the initial dataset already had 80 features to work with. \n- **EDA** - We explore some of the more important attributes of the dataset.\n- **Feature Scaling** - Examining the differing scaling techniques and the effect it has on the data\n- **Dimensionality Reduction** - Used PCA to determine which scaling/transforming/normalizing method had the lowest number of components for the highest possible variance.\n- **Feature Importance** - Using RandomForest & LassoCV to determine which features are most important.\n- **Modelling** - Ensembling with a Stacking Classifer. Also a neural network was used.\n- **Assessing The Models** - A variety of metrics were used to evaluate model performance. ","metadata":{}},{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport plotly\nimport plotly.express as px\nfrom plotly.subplots import make_subplots\nimport plotly.graph_objects as go\n\n\nfrom plotly.offline import plot, iplot, init_notebook_mode\ninit_notebook_mode(connected=True)\nimport matplotlib.pyplot as plt\n\nimport warnings\nwarnings.filterwarnings('ignore')\n\nRANDOM_STATE = 115","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-07-25T23:56:47.801682Z","iopub.execute_input":"2022-07-25T23:56:47.802140Z","iopub.status.idle":"2022-07-25T23:56:49.283838Z","shell.execute_reply.started":"2022-07-25T23:56:47.802046Z","shell.execute_reply":"2022-07-25T23:56:49.282621Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = pd.read_csv('../input/house-prices-advanced-regression-techniques/train.csv')\ndf_train.head()","metadata":{"_kg_hide-input":false,"_kg_hide-output":true,"execution":{"iopub.status.busy":"2022-07-25T23:56:49.285563Z","iopub.execute_input":"2022-07-25T23:56:49.285979Z","iopub.status.idle":"2022-07-25T23:56:49.359262Z","shell.execute_reply.started":"2022-07-25T23:56:49.285946Z","shell.execute_reply":"2022-07-25T23:56:49.356986Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test = pd.read_csv('../input/house-prices-advanced-regression-techniques/test.csv')\ndf_test.head()","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2022-07-25T23:56:49.360780Z","iopub.execute_input":"2022-07-25T23:56:49.361243Z","iopub.status.idle":"2022-07-25T23:56:49.426720Z","shell.execute_reply.started":"2022-07-25T23:56:49.361201Z","shell.execute_reply":"2022-07-25T23:56:49.425926Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def print_unique_values_df(df: pd.DataFrame):\n    for col in list(df):\n        print(\"Unique Values for \"'{}'\":{}\".format(str(col), df[col].unique()))\n        print(\"Num Unique Values for \"'{}'\":{} values \".format(str(col), df[col].nunique()))\n        print(\"dtype for {} is :{}\".format(str(col), df[col].dtypes))\n        print(\"-\" * 150)\n\nprint_unique_values_df(df_train)","metadata":{"_kg_hide-output":true,"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-07-25T23:56:49.429340Z","iopub.execute_input":"2022-07-25T23:56:49.430344Z","iopub.status.idle":"2022-07-25T23:56:49.507417Z","shell.execute_reply.started":"2022-07-25T23:56:49.430299Z","shell.execute_reply":"2022-07-25T23:56:49.506355Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_percentage_missing(df:pd.DataFrame,sort_val:bool = True,ascending:bool=False,col_name:str=\"percent_missing\"):\n    if sort_val == True :\n        return pd.DataFrame(df.isnull().sum().sort_values(ascending=ascending)/len(df)*100,columns=[col_name])  \n    elif sort_val == False:\n        return pd.DataFrame(df.isnull().sum()/len(df)*100,columns=[col_name])  ","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-07-25T23:56:49.508698Z","iopub.execute_input":"2022-07-25T23:56:49.509026Z","iopub.status.idle":"2022-07-25T23:56:49.515684Z","shell.execute_reply.started":"2022-07-25T23:56:49.508996Z","shell.execute_reply":"2022-07-25T23:56:49.514572Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def drop_cols_threshold_greater_than(df:pd.DataFrame,threshold:float = 50,geq:bool=True):\n    return df.drop(cols_with_nan_threshold_greater_than(df,threshold,geq=geq),axis=1)\n    \ndef drop_cols_threshold_less_than(df:pd.DataFrame,threshold:float = 50,leq:bool=True):\n    return df.drop(cols_with_nan_threshold_less_than(df,threshold,leq=leq),axis=1)\n\ndef cols_with_nan_threshold_greater_than(df:pd.DataFrame,threshold:float = 50,geq:bool = True):\n    check_threshold(threshold)\n    if geq:\n        threshold_cols = [col for col in list(df.columns) if df[col].isnull().sum()/len(df)*100 >= threshold]\n    elif not geq:\n        threshold_cols = [col for col in list(df.columns) if df[col].isnull().sum()/len(df)*100 > threshold]\n    return threshold_cols \n\ndef cols_with_nan_threshold_less_than(df:pd.DataFrame,threshold:float = 50,leq:bool = True):\n    check_threshold(threshold)\n    if leq:\n        threshold_cols = [col for col in list(df.columns) if df[col].isnull().sum()/len(df)*100 <= threshold]\n    elif not leq:\n        threshold_cols = [col for col in list(df.columns) if df[col].isnull().sum()/len(df)*100 < threshold]\n    return threshold_cols \n\ndef check_threshold(threshold:float):\n    if threshold < 0:\n        raise ValueError(\"Threshold cannot be less than 0\")\n    if threshold > 100:\n        raise ValueError(\"Threshold cannot be more than 100\")","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-07-25T23:56:49.517346Z","iopub.execute_input":"2022-07-25T23:56:49.517773Z","iopub.status.idle":"2022-07-25T23:56:49.531079Z","shell.execute_reply.started":"2022-07-25T23:56:49.517742Z","shell.execute_reply":"2022-07-25T23:56:49.529714Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def show_percentage_missing(df:pd.DataFrame,\n                            sort_val:bool = True,\n                            ascending:bool= False,\n                            color:str = \"firebrick\",\n                            col_name = \"percent_missing\",\n                            title:str = \"Percentage of data missing.\",\n                            opacity:float=1.0):\n    if opacity <0 or opacity > 1:\n        raise ValueError(\"Invalid Opacity\")\n    df = get_percentage_missing(df,sort_val,ascending,col_name)\n    fig = px.histogram(df,\n                   x=col_name,\n                   y=df.index,\n                   color_discrete_sequence=[color],\n                   opacity = opacity)\n    fig.update_xaxes(title=\"Percentage (%) Missing\")\n    fig.update_yaxes(title=\"Column Name\")\n    fig.update_layout(title=title)\n    fig.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-07-25T23:56:49.532942Z","iopub.execute_input":"2022-07-25T23:56:49.533436Z","iopub.status.idle":"2022-07-25T23:56:49.545645Z","shell.execute_reply.started":"2022-07-25T23:56:49.533376Z","shell.execute_reply":"2022-07-25T23:56:49.544579Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Check to see if columns are the same in train and test sets, sale price is missing from the test set, however this is the target variable, so no issues here","metadata":{}},{"cell_type":"code","source":"print(f\"Are the columns in the train and test the same?: {set(list(df_train.columns))==set(list(df_test.columns))}\")\ntrain_set_cols = set(list(df_train.columns))\ntest_set_cols = set(list(df_test.columns))\nprint(f\"Difference in columns: {train_set_cols.difference(test_set_cols)} \")","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-07-25T23:56:49.547234Z","iopub.execute_input":"2022-07-25T23:56:49.547953Z","iopub.status.idle":"2022-07-25T23:56:49.561004Z","shell.execute_reply.started":"2022-07-25T23:56:49.547908Z","shell.execute_reply":"2022-07-25T23:56:49.560230Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Missing Data in the Train & Test Sets","metadata":{}},{"cell_type":"code","source":"show_percentage_missing(df=df_train,ascending=False,sort_val = True,title = \"Percentage of data missing (Train)\",color=\"purple\")\nshow_percentage_missing(df=df_test,ascending=False,sort_val = True,title = \"Percentage of data missing (Test)\",color=\"goldenrod\")","metadata":{"execution":{"iopub.status.busy":"2022-07-25T23:56:49.562329Z","iopub.execute_input":"2022-07-25T23:56:49.562641Z","iopub.status.idle":"2022-07-25T23:56:50.559482Z","shell.execute_reply.started":"2022-07-25T23:56:49.562612Z","shell.execute_reply":"2022-07-25T23:56:50.558373Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train_numeric = df_train.select_dtypes(exclude=\"object\")\n# show_percentage_missing(df_train_numeric,color=\"darkorchid\",title=\"Percentage of data missing in numeric features (Train)\",opacity = 0.9)\n\ndf_test_numeric = df_test.select_dtypes(exclude=\"object\")\n# show_percentage_missing(df_test_numeric,color=\"darkblue\",title=\"Percentage of data missing in numeric features (Test)\",opacity = 0.9)","metadata":{"execution":{"iopub.status.busy":"2022-07-25T23:56:50.564978Z","iopub.execute_input":"2022-07-25T23:56:50.565352Z","iopub.status.idle":"2022-07-25T23:56:50.576543Z","shell.execute_reply.started":"2022-07-25T23:56:50.565321Z","shell.execute_reply":"2022-07-25T23:56:50.575189Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Dealing with missing data","metadata":{}},{"cell_type":"markdown","source":"## Numerical","metadata":{}},{"cell_type":"markdown","source":"Consider dealing with lotfrontage a different way, as it's approximately 17% missing data in both sets","metadata":{}},{"cell_type":"code","source":"dummy_col = np.random.random(len(df_test_numeric))\ndf_test_numeric = pd.concat([df_test_numeric,pd.Series(dummy_col).rename(\"SalePrice\")],axis = 1)","metadata":{"execution":{"iopub.status.busy":"2022-07-25T23:56:50.578519Z","iopub.execute_input":"2022-07-25T23:56:50.578947Z","iopub.status.idle":"2022-07-25T23:56:50.596013Z","shell.execute_reply.started":"2022-07-25T23:56:50.578905Z","shell.execute_reply":"2022-07-25T23:56:50.594986Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Simple Imputer \n\nImputation transformer for completing missing values.","metadata":{}},{"cell_type":"code","source":"from sklearn.impute import SimpleImputer","metadata":{"execution":{"iopub.status.busy":"2022-07-25T23:56:50.597643Z","iopub.execute_input":"2022-07-25T23:56:50.598792Z","iopub.status.idle":"2022-07-25T23:56:51.636232Z","shell.execute_reply.started":"2022-07-25T23:56:50.598748Z","shell.execute_reply":"2022-07-25T23:56:51.635029Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"old_lot_frontage = df_train_numeric[\"LotFrontage\"]","metadata":{"execution":{"iopub.status.busy":"2022-07-25T23:56:51.637885Z","iopub.execute_input":"2022-07-25T23:56:51.638393Z","iopub.status.idle":"2022-07-25T23:56:51.644434Z","shell.execute_reply.started":"2022-07-25T23:56:51.638345Z","shell.execute_reply":"2022-07-25T23:56:51.643037Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"old_df_train_numeric = df_train_numeric\nold_df_test_numeric = df_test_numeric","metadata":{"execution":{"iopub.status.busy":"2022-07-25T23:56:51.645837Z","iopub.execute_input":"2022-07-25T23:56:51.646441Z","iopub.status.idle":"2022-07-25T23:56:51.660726Z","shell.execute_reply.started":"2022-07-25T23:56:51.646401Z","shell.execute_reply":"2022-07-25T23:56:51.659610Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"imputer = SimpleImputer(strategy=\"median\")\nX_train = imputer.fit_transform(df_train_numeric)\ndf_train_numeric = pd.DataFrame(X_train,columns = list(df_train_numeric.columns))\n\nX_test = imputer.transform(df_test_numeric)\ndf_test_numeric = pd.DataFrame(X_test,columns = list(df_test_numeric.columns))\ndf_test_numeric = df_test_numeric.drop([\"SalePrice\"],axis = 1)","metadata":{"execution":{"iopub.status.busy":"2022-07-25T23:56:51.662035Z","iopub.execute_input":"2022-07-25T23:56:51.662585Z","iopub.status.idle":"2022-07-25T23:56:51.689913Z","shell.execute_reply.started":"2022-07-25T23:56:51.662550Z","shell.execute_reply":"2022-07-25T23:56:51.688816Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = make_subplots(\n    rows=1, cols=2, subplot_titles=(\"Before Imputation\", \"Simple Imputation (median)\")\n)\n\nfig.add_trace(go.Histogram(x=old_lot_frontage,showlegend=False,marker_color='#EB89B5'), row=1, col=1)\nfig.add_trace(go.Histogram(x=df_train_numeric[\"LotFrontage\"],showlegend=False,marker_color='#1E90ff',opacity=0.9), row=1, col=2)\n\nfig.update_xaxes(title_text=\"Lot Frontage\", row=1, col=1)\nfig.update_xaxes(title_text=\"Lot Frontage\",  row=1, col=2)\nfig.update_yaxes(title_text=\"Count\", row=1, col=1)\nfig.update_yaxes(title_text=\"Count\",  row=1, col=2)\nfig.update_layout(title_text=\"Lot Frontage Distributions\")\nfig.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-07-25T23:56:51.691531Z","iopub.execute_input":"2022-07-25T23:56:51.692237Z","iopub.status.idle":"2022-07-25T23:56:51.754723Z","shell.execute_reply.started":"2022-07-25T23:56:51.692192Z","shell.execute_reply":"2022-07-25T23:56:51.753572Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### KNN Imputer\n\nImputation for completing missing values using k-Nearest Neighbors.","metadata":{}},{"cell_type":"code","source":"from sklearn.impute import KNNImputer","metadata":{"execution":{"iopub.status.busy":"2022-07-25T23:56:51.756153Z","iopub.execute_input":"2022-07-25T23:56:51.756750Z","iopub.status.idle":"2022-07-25T23:56:51.760573Z","shell.execute_reply.started":"2022-07-25T23:56:51.756717Z","shell.execute_reply":"2022-07-25T23:56:51.759843Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"imputer = KNNImputer(n_neighbors=5)\n\nX_train = imputer.fit_transform(old_df_train_numeric)\ndf_train_numeric = pd.DataFrame(X_train,columns = list(old_df_train_numeric.columns))\n\nX_test = imputer.transform(old_df_test_numeric)\ndf_test_numeric = pd.DataFrame(X_test,columns = list(old_df_test_numeric.columns))\ndf_test_numeric = df_test_numeric.drop([\"SalePrice\"],axis = 1)","metadata":{"execution":{"iopub.status.busy":"2022-07-25T23:56:51.761890Z","iopub.execute_input":"2022-07-25T23:56:51.762440Z","iopub.status.idle":"2022-07-25T23:56:51.998369Z","shell.execute_reply.started":"2022-07-25T23:56:51.762407Z","shell.execute_reply":"2022-07-25T23:56:51.996995Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = make_subplots(\n    rows=1, cols=2, subplot_titles=(\"Before Imputation\", \"KNN Imputation (neighbors = 5)\")\n)\n\nfig.add_trace(go.Histogram(x=old_lot_frontage,showlegend=False,marker_color='#EB89B5'), row=1, col=1)\nfig.add_trace(go.Histogram(x=df_train_numeric[\"LotFrontage\"],showlegend=False,marker_color='#1E90ff',opacity=0.9), row=1, col=2)\n\nfig.update_xaxes(title_text=\"Lot Frontage\", row=1, col=1)\nfig.update_xaxes(title_text=\"Lot Frontage\",  row=1, col=2)\nfig.update_yaxes(title_text=\"Count\", row=1, col=1)\nfig.update_yaxes(title_text=\"Count\",  row=1, col=2)\nfig.update_layout(title_text=\"Lot Frontage Distributions\")\nfig.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-07-25T23:56:52.000102Z","iopub.execute_input":"2022-07-25T23:56:52.001043Z","iopub.status.idle":"2022-07-25T23:56:52.069960Z","shell.execute_reply.started":"2022-07-25T23:56:52.000997Z","shell.execute_reply":"2022-07-25T23:56:52.068872Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Categorical","metadata":{}},{"cell_type":"code","source":"df_train_categorical = df_train.select_dtypes(exclude=[\"float\",\"int\"])\n# show_percentage_missing(df_train_categorical,color = \"mediumorchid\",title=\"Percentage of Cateogorical data missing (Train)\")\n\ndf_test_categorical = df_test.select_dtypes(exclude=[\"float\",\"int\"])\n# show_percentage_missing(df_test_categorical,color = \"blue\",title=\"Percentage of Cateogorical data missing (Test)\")","metadata":{"execution":{"iopub.status.busy":"2022-07-25T23:56:52.071658Z","iopub.execute_input":"2022-07-25T23:56:52.072581Z","iopub.status.idle":"2022-07-25T23:56:52.082902Z","shell.execute_reply.started":"2022-07-25T23:56:52.072536Z","shell.execute_reply":"2022-07-25T23:56:52.081771Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Drop columns with a missing percentage of 45% ","metadata":{}},{"cell_type":"code","source":"threshold:float = 45\n    \nprint(f\"Categorical Columns With A Threshold of nan's greater than {threshold} % is {cols_with_nan_threshold_greater_than(df_train_categorical,threshold=threshold)} in Train set\")\n\ntrain_set_cols_drop = cols_with_nan_threshold_greater_than(df_train_categorical,threshold=threshold)\n\ndf_train_categorical = drop_cols_threshold_greater_than(df_train_categorical,threshold = threshold,geq=True)\ndf_train_categorical= df_train_categorical.apply(lambda x:x.fillna(x.value_counts().index[0]))\n\ndf_test_categorical = df_test_categorical.drop(train_set_cols_drop,axis = 1) \ndf_test_categorical= df_test_categorical.apply(lambda x:x.fillna(x.value_counts().index[0]))\n\ndf_train_categorical = df_train_categorical[list(df_train_categorical.columns)].astype(\"category\")\ndf_test_categorical = df_test_categorical[list(df_test_categorical.columns)].astype(\"category\")","metadata":{"execution":{"iopub.status.busy":"2022-07-25T23:56:52.084065Z","iopub.execute_input":"2022-07-25T23:56:52.084966Z","iopub.status.idle":"2022-07-25T23:56:52.254228Z","shell.execute_reply.started":"2022-07-25T23:56:52.084931Z","shell.execute_reply":"2022-07-25T23:56:52.253003Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train_clean = pd.concat([df_train_numeric,df_train_categorical],axis = 1)\ndf_test_clean = pd.concat([df_test_numeric,df_test_categorical],axis = 1)","metadata":{"execution":{"iopub.status.busy":"2022-07-25T23:56:52.255934Z","iopub.execute_input":"2022-07-25T23:56:52.256374Z","iopub.status.idle":"2022-07-25T23:56:52.264703Z","shell.execute_reply.started":"2022-07-25T23:56:52.256342Z","shell.execute_reply":"2022-07-25T23:56:52.263668Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Binning Values\n\nBinning is a way to group a number of more or less continuous values into a smaller number of \"bins\". It also makes visualization of these features easier. ","metadata":{}},{"cell_type":"markdown","source":"### Binning Years into Decades","metadata":{}},{"cell_type":"code","source":"decade_bins = [1799]\ndecade_labels = []\nwhile decade_bins[-1] < 2009:\n    decade_labels.append(str(decade_bins[-1]+1)+\"'s\")\n    decade_bins.append(decade_bins[-1]+10)\ndecade_bins.append(decade_bins[-1]+1)\ndecade_labels.append(str(decade_bins[-1])+\"'s\")\ndf_train_clean[\"decade\"] = pd.cut(x=df_train_clean['YearBuilt'], bins=decade_bins,labels=decade_labels)","metadata":{"execution":{"iopub.status.busy":"2022-07-25T23:56:52.265993Z","iopub.execute_input":"2022-07-25T23:56:52.266392Z","iopub.status.idle":"2022-07-25T23:56:52.282286Z","shell.execute_reply.started":"2022-07-25T23:56:52.266361Z","shell.execute_reply":"2022-07-25T23:56:52.281142Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Binning Years into Quarter Centuries","metadata":{}},{"cell_type":"code","source":"quarter_century_bins = [1799]\nquarter_century_labels= []\nwhile quarter_century_bins[-1] < 2000:\n    quarter_century_labels.append(str(quarter_century_bins[-1]+1)+\"'s\")\n    quarter_century_bins.append(quarter_century_bins[-1]+25)\n    \ndf_train_clean[\"quarter_century\"] = pd.cut(x=df_train_clean['YearBuilt'], bins=quarter_century_bins,labels=quarter_century_labels)","metadata":{"execution":{"iopub.status.busy":"2022-07-25T23:56:52.284252Z","iopub.execute_input":"2022-07-25T23:56:52.285062Z","iopub.status.idle":"2022-07-25T23:56:52.295937Z","shell.execute_reply.started":"2022-07-25T23:56:52.285016Z","shell.execute_reply":"2022-07-25T23:56:52.294713Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Binning Years into Half Centuries","metadata":{}},{"cell_type":"code","source":"half_century_bins = [1799]\nhalf_century_labels= []\nwhile half_century_bins[-1] < 2000:\n    half_century_labels.append(str(half_century_bins[-1]+1)+\"'s\")\n    half_century_bins.append(half_century_bins[-1]+50)\n    \ndf_train_clean[\"half_century\"] = pd.cut(x=df_train_clean['YearBuilt'], bins=half_century_bins,labels=half_century_labels)","metadata":{"execution":{"iopub.status.busy":"2022-07-25T23:56:52.298030Z","iopub.execute_input":"2022-07-25T23:56:52.298418Z","iopub.status.idle":"2022-07-25T23:56:52.310481Z","shell.execute_reply.started":"2022-07-25T23:56:52.298386Z","shell.execute_reply":"2022-07-25T23:56:52.309598Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Binning Years into Centuries","metadata":{}},{"cell_type":"code","source":"df_train_clean['century'] = pd.cut(x=df_train_clean['YearBuilt'], bins=[1799,1899,1999,2010],labels=[\"19th Century\",\"20th Century\",\"21st Century\"])","metadata":{"execution":{"iopub.status.busy":"2022-07-25T23:56:52.311536Z","iopub.execute_input":"2022-07-25T23:56:52.312315Z","iopub.status.idle":"2022-07-25T23:56:52.325083Z","shell.execute_reply.started":"2022-07-25T23:56:52.312282Z","shell.execute_reply":"2022-07-25T23:56:52.324058Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Binning Overall Quality & Sale Price","metadata":{}},{"cell_type":"code","source":"df_train_clean[\"qual_metric\"] = pd.cut(x=df_train_clean['OverallQual'], bins=[0.0,3.0,5.0,8.0,10.0],labels=[\"Worst\",\"Bad\",\"Good\",\"Excellent\"])\ndf_train_clean[\"SalePriceBinned\"] = pd.cut(x=df_train_clean['SalePrice'], bins=[0.0,99999.99, 199999.99,299999.99,399999.99,499999.99,599999.99,699999.99,799999.99,899999.99,1000000.00],\n                                           labels=[\"<100K\",\"100k - 199K\",\"200k - 299K\",\"300k - 399K\",\"400k - 499K\",\"500k - 599K\",\"600k - 699K\",\"700k - 799K\",\"800k - 899K\",\"900k - 1M\",])","metadata":{"execution":{"iopub.status.busy":"2022-07-25T23:56:52.326342Z","iopub.execute_input":"2022-07-25T23:56:52.326833Z","iopub.status.idle":"2022-07-25T23:56:52.342716Z","shell.execute_reply.started":"2022-07-25T23:56:52.326804Z","shell.execute_reply":"2022-07-25T23:56:52.341397Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Exploratory Data Analysis","metadata":{}},{"cell_type":"markdown","source":"## Pairplot\nPair plot of the features with the top four variances","metadata":{}},{"cell_type":"code","source":"# n = 4\n# highest_n_variances = df_train_clean.var().sort_values(ascending=not True)[0:n]\n\n# fig = px.scatter_matrix(df_train_clean[list(highest_n_variances.index)],opacity=0.2)\n# fig.update_traces(diagonal_visible=False)\n# iplot(fig)","metadata":{"execution":{"iopub.status.busy":"2022-07-25T23:56:52.350575Z","iopub.execute_input":"2022-07-25T23:56:52.351106Z","iopub.status.idle":"2022-07-25T23:56:52.354624Z","shell.execute_reply.started":"2022-07-25T23:56:52.351075Z","shell.execute_reply":"2022-07-25T23:56:52.353904Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Correlation\n\nA good starting point, is to check which features are most correlated with our target variable (SalePrice)\n\nThe features that are most correlated with Sale Price are: \n\n*  'OverallQual'; I mean, this makes sense, houses that are of better quality may have cost more overall to be built, therefore a higher cost.\n*  'GrLivArea' 'TotalBsmtSF', '1stFlrSF'; These seem to be related, an increase in space would lead to an increase in SalePrice\n*  'GarageCars' & 'GarageArea'; can be thought of the same, as you would need more space for more cars. \n*  'FullBath'; Seems people pay would pay a premium for their own private bathrooms. \n*  'TotRmsAbvGrd'\n*  'YearBuilt', 'YearRemodAdd'; \n\nApart from the Year Built and the Year a Remodel was done, all the highly correleated features seem to be associated with space, i.e, *more space means a higher sale price*.","metadata":{}},{"cell_type":"code","source":"corr = df_train_clean.corr()\nx = corr.nlargest(10+1,\"SalePrice\").index #+1 because sale price is correlated with itself \ncorr_df = df_train_clean[list(x)]\ncorr = corr_df.corr()\nfig = px.imshow(corr)\nfig.update_layout(title=\"Top 10 Features Correlated With Sale Price\")\nfig.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-07-25T23:56:52.355818Z","iopub.execute_input":"2022-07-25T23:56:52.356151Z","iopub.status.idle":"2022-07-25T23:56:52.446695Z","shell.execute_reply.started":"2022-07-25T23:56:52.356121Z","shell.execute_reply":"2022-07-25T23:56:52.445588Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Removing the outliers of Sale Price.","metadata":{}},{"cell_type":"code","source":"sale_price = df_train_clean[\"SalePrice\"]\niqr = np.quantile(sale_price,0.75)-np.quantile(sale_price,0.25)\nlower = np.quantile(sale_price,0.25)-1.5*iqr\nupper = np.quantile(sale_price,0.75)+1.5*iqr\ndf_train_no_outliers = df_train_clean[(df_train_clean[\"SalePrice\"]>lower)&(df_train_clean[\"SalePrice\"]<upper)]","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-07-25T23:56:52.448043Z","iopub.execute_input":"2022-07-25T23:56:52.448877Z","iopub.status.idle":"2022-07-25T23:56:52.459890Z","shell.execute_reply.started":"2022-07-25T23:56:52.448845Z","shell.execute_reply":"2022-07-25T23:56:52.458742Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Kurtosis & Skewness","metadata":{}},{"cell_type":"code","source":"print(f' Sale Price Skewness: {df_train_clean[\"SalePrice\"].skew()}')\nprint(f' Sale Price Kurtosis: {df_train_clean[\"SalePrice\"].kurtosis()}')\nprint(f' Sale Price (Outliers Removed) Skewness: {df_train_no_outliers[\"SalePrice\"].skew()}')\nprint(f' Sale Price (Outliers Removed) Kurtosis: {df_train_no_outliers[\"SalePrice\"].kurtosis()}')","metadata":{"execution":{"iopub.status.busy":"2022-07-25T23:56:52.461491Z","iopub.execute_input":"2022-07-25T23:56:52.461860Z","iopub.status.idle":"2022-07-25T23:56:52.469351Z","shell.execute_reply.started":"2022-07-25T23:56:52.461832Z","shell.execute_reply":"2022-07-25T23:56:52.468334Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Shapiro - Wilk Test For Normality","metadata":{}},{"cell_type":"code","source":"from scipy.stats import shapiro\n\nprint(f' Sale Price (p-value)): {shapiro(df_train_clean[\"SalePrice\"]).pvalue}')\nprint(f' Sale Price (Outliers Removed) (p-value): {shapiro(df_train_no_outliers[\"SalePrice\"]).pvalue}')\n\nprint(f' Sale Price (Log Transformed)(p-value)): {shapiro(np.log(df_train_clean[\"SalePrice\"])).pvalue}')\nprint(f' Sale Price (Log Transformed) (Outliers Removed) (p-value): {shapiro(np.log(df_train_no_outliers[\"SalePrice\"])).pvalue}')\n\nprint(f' Sale Price (Square Root Transformed)(p-value)): {shapiro(np.power(df_train_clean[\"SalePrice\"],1/2)).pvalue}')\nprint(f' Sale Price (Square Root Transformed) (Outliers Removed) (p-value): {shapiro(np.power(df_train_no_outliers[\"SalePrice\"],1/2)).pvalue}')\n\nprint(f' Sale Price (Cube Root Transformed)(p-value)): {shapiro(np.power(df_train_clean[\"SalePrice\"],1/3)).pvalue}')\nprint(f' Sale Price (Cube Root Transformed) (Outliers Removed) (p-value): {shapiro(np.power(df_train_no_outliers[\"SalePrice\"],1/3)).pvalue}')\n\nprint(f' Sale Price (Cube Root Transformed)(p-value)): {shapiro(np.power(df_train_clean[\"SalePrice\"],1/4)).pvalue}')\nprint(f' Sale Price (Cube Root Transformed) (Outliers Removed) (p-value): {shapiro(np.power(df_train_no_outliers[\"SalePrice\"],1/4)).pvalue}')","metadata":{"execution":{"iopub.status.busy":"2022-07-25T23:56:52.470817Z","iopub.execute_input":"2022-07-25T23:56:52.471132Z","iopub.status.idle":"2022-07-25T23:56:52.483386Z","shell.execute_reply.started":"2022-07-25T23:56:52.471103Z","shell.execute_reply":"2022-07-25T23:56:52.482138Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Examining The Distribution of the Target\n\nWe examine the SalePrice as it is and then with the removal of the outliers.\n\nSkewness assesses the extent to which a variable’s distribution is symmetrical. \nIf the distribution of responses for a variable stretches toward the right or left tail of the distribution, then the distribution is referred to as skewed.\n\nKurtosis is a measure of whether the distribution is too peaked (a very narrow distribution with most of the responses in the center).\n\nWhen both skewness and kurtosis are zero (a situation that researchers are very unlikely to ever encounter), the pattern of responses is considered a normal distribution.\n\nA general guideline for skewness is that if the number is greater than +1 or lower than –1, this is an indication of a substantially skewed distribution. For kurtosis, the general guideline is that if the number is greater than +1, the distribution is too peaked. Likewise, a kurtosis of less than –1 indicates a distribution that is too flat. Distributions exhibiting skewness and/or kurtosis that exceed these guidelines are considered nonnormal. **[Src](https://www.smartpls.com/documentation/functionalities/excess-kurtosis-and-skewness/)**\n\n* **Sale Price Skewness** $\\approx$ 1.8\n* **Sale Price Kurtosis** $\\approx$ 6.5\n* **Sale Price (Outliers Removed) Skewness** $\\approx$ 0.68\n* **Sale Price (Outliers Removed) Kurtosis** $\\approx$ 0.091\n\nWe can also use the **Shapiro - Wilk Test** for test of normality.\n\n*  **Sale Price (p-value)**: 3.206247534576162e-33\n\nSince the p-value is less than .05, we have sufficient evidence to say that the data does not come from a normal distribution.\n*  **Sale Price (Outliers Removed) (p-value)**: 1.53821483074353e-18\n\nSince the p-value is less than .05, we have sufficient evidence to say that the data does not come from a normal distribution.\n\n*  **Sale Price (Log Transformed)(p-value))**: 1.1490678986092462e-07\n*  **Sale Price (Log Transformed) (Outliers Removed) (p-value)**: 7.837464566229357e-10\n*  **Sale Price (Square Root Transformed)(p-value))**: 1.1143702087512663e-20\n*  **Sale Price (Square Root Transformed) (Outliers Removed) (p-value)**: 1.0709845454925926e-08\n*  **Sale Price (Cube Root Transformed)(p-value))**: 9.70732262442779e-16\n*  **Sale Price (Cube Root Transformed) (Outliers Removed) (p-value)**: 1.0642602319421712e-06\n*  **Sale Price (Cube Root Transformed)(p-value))**: 3.0327690771041194e-13\n*  **Sale Price (Cube Root Transformed) (Outliers Removed) (p-value)**: 1.5475717418667045e-06\n\nSince the p-value is less than .05 for all the transformations, we have sufficient evidence to say that the data does not come from a normal distribution for each transformation.","metadata":{}},{"cell_type":"code","source":"fig = make_subplots(\n    rows=4, cols=2, subplot_titles=(\"Original\", \"With no Outliers\",\n                                    \"Original (Log Transformed)\", \"With no Outliers (Log Transformed)\",\n                                    \"Original (sqrt Transformed)\", \"With no Outliers (sqrt Transformed)\",\n                                   \"Original (cube rt Transformed)\", \"With no Outliers (cube rt Transformed)\")\n)\n\nfig.add_trace(go.Histogram(x=df_train_clean[\"SalePrice\"],showlegend=False,marker_color='#EB89B5'), row=1, col=1)\nfig.add_trace(go.Histogram(x=df_train_no_outliers[\"SalePrice\"],showlegend=False,marker_color='#1E90ff',opacity=0.9), row=1, col=2)\n\nfig.add_trace(go.Histogram(x=np.log(df_train_clean[\"SalePrice\"]),showlegend=False,marker_color='#EB89B5'), row=2, col=1)\nfig.add_trace(go.Histogram(x=np.log(df_train_no_outliers[\"SalePrice\"]),showlegend=False,marker_color='#1E90ff',opacity=0.9), row=2, col=2)\n\nfig.add_trace(go.Histogram(x=np.power(df_train_clean[\"SalePrice\"],1/2),showlegend=False,marker_color='#EB89B5'), row=3, col=1)\nfig.add_trace(go.Histogram(x=np.power(df_train_no_outliers[\"SalePrice\"],1/2),showlegend=False,marker_color='#1E90ff',opacity=0.9), row=3, col=2)\n\nfig.add_trace(go.Histogram(x=np.power(df_train_clean[\"SalePrice\"],1/3),showlegend=False,marker_color='#EB89B5'), row=4, col=1)\nfig.add_trace(go.Histogram(x=np.power(df_train_no_outliers[\"SalePrice\"],1/3),showlegend=False,marker_color='#1E90ff',opacity=0.9), row=4, col=2)\n\n\nfig.update_xaxes(title_text=\"Sale Price\", row=1, col=1)\nfig.update_xaxes(title_text=\"Sale Price\",  row=1, col=2)\nfig.update_yaxes(title_text=\"Count\", row=1, col=1)\nfig.update_yaxes(title_text=\"Count\",  row=1, col=2)\nfig.update_layout(title_text=\"Sale Price Distributions\",height=1200)\nfig.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-07-25T23:56:52.485015Z","iopub.execute_input":"2022-07-25T23:56:52.485617Z","iopub.status.idle":"2022-07-25T23:56:52.633774Z","shell.execute_reply.started":"2022-07-25T23:56:52.485583Z","shell.execute_reply":"2022-07-25T23:56:52.632670Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## ECDF Plot of Sale Price \n\n*ECDF*:Empirical Cumulative Distribution Function\n\n* From the first graph, we can see that 99% of sale prices are equal to or below $446,000.00$\n\n* From the second graph, we can see houses that are of \"Excellent\" quality, 20% of it's sale prices are greater than $452,000.00$,\n* From the second graph, we can see houses that are of \"Good\" quality, 99% of it's sale prices are equal to or below $403,000.00$\n* From the second graph, we can see houses that are of \"Bad\" quality, 99% of it's sale prices are equal to or below $228,000.00$\n* From the second graph, we can see houses that are of \"Worst\" quality, 92% of it's sale prices are equal to or below $120,000.00$","metadata":{}},{"cell_type":"code","source":"df_train_clean = df_train_clean.sort_values(by=\"OverallQual\",ascending=True)\n\nfig = px.ecdf(df_train_clean,x=\"SalePrice\",color_discrete_sequence=[\"deeppink\"])\nfig.update_layout(title=\"ECDF Of Sale Price\")\n\nfig.add_vline(x=np.quantile(df_train_clean[\"SalePrice\"], 0.1), line_width=3, line_dash=\"dash\", line_color=\"blue\",annotation_text=str(np.quantile(df_train_clean[\"SalePrice\"], 0.1))+\"K\")\n# fig.add_vline(x=100000, line_width=3, line_dash=\"dash\", line_color=\"blue\",annotation_text=\"100K\")\nfig.add_vline(x=300000, line_width=3, line_dash=\"dash\", line_color=\"blue\",annotation_text=\"300K\")\nfig.add_vline(x=400000, line_width=3, line_dash=\"dash\", line_color=\"blue\",annotation_text=\"400K\")\nfig.add_vline(x=446261, line_width=3, line_dash=\"dash\", line_color=\"blue\",annotation_text=\"446K\")\nfig.add_vline(x=500000, line_width=3, line_dash=\"dash\", line_color=\"blue\",annotation_text=\"500K\")\nfig.add_vline(x=600000, line_width=3, line_dash=\"dash\", line_color=\"blue\",annotation_text=\"600K\")\nfig.add_vline(x=np.quantile(df_train_clean[\"SalePrice\"], 0.999315),\n              line_width=3, line_dash=\"dash\", \n              line_color=\"orange\",\n              annotation_text=str(np.floor(np.quantile(df_train_clean[\"SalePrice\"], 0.999315)))+\"K\")\nfig.add_vline(x=np.median(df_train_clean[\"SalePrice\"]),line_dash=\"dot\",\n              line_color=\"green\",\n              line_width=5,\n              annotation_text = str(np.floor(np.median(df_train_clean[\"SalePrice\"])))+\"K (median)\")\niplot(fig)","metadata":{"_kg_hide-input":false,"execution":{"iopub.status.busy":"2022-07-25T23:56:52.634982Z","iopub.execute_input":"2022-07-25T23:56:52.635430Z","iopub.status.idle":"2022-07-25T23:56:52.923430Z","shell.execute_reply.started":"2022-07-25T23:56:52.635373Z","shell.execute_reply":"2022-07-25T23:56:52.922097Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = px.ecdf(df_train_clean,x=\"SalePrice\",color=\"qual_metric\",color_discrete_sequence=[\"darkred\",\"goldenrod\",\"green\",\"dodgerblue\"])\nfig.update_layout(title=\"ECDF Of Sale Price accoring to quality\")\n# fig.add_vrect(x0=np.quantile(df_train_clean[\"SalePrice\"], 0.1), x1=np.quantile(df_train_clean[\"SalePrice\"], 0.9))\niplot(fig)","metadata":{"execution":{"iopub.status.busy":"2022-07-25T23:56:52.924815Z","iopub.execute_input":"2022-07-25T23:56:52.925243Z","iopub.status.idle":"2022-07-25T23:56:53.050146Z","shell.execute_reply.started":"2022-07-25T23:56:52.925201Z","shell.execute_reply":"2022-07-25T23:56:53.048968Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Sale Price Distributions in Different Timeframes","metadata":{}},{"cell_type":"markdown","source":"### Checking out the amount of houses that were built per decade\n* It looks like the most houses in the data were built in the 2000's.\n* 1950's, 1960's, 1970's & 1990's all seem to have a similar built amount.\n* 1900's and before have a small build amoumt, less than 20 houses per decade in the data were built then.","metadata":{}},{"cell_type":"code","source":"fig = px.bar(df_train_clean.groupby(\"decade\")[[\"SalePrice\"]].count(),color_discrete_sequence=[\"dodgerblue\"])\nfig.update_layout(showlegend=False)\nfig.update_xaxes(title=\"Decade Built\")\nfig.update_xaxes(title=\"Count\")\nfig.update_layout(title = \"Count of houses built per decade\")\nfig.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-07-25T23:56:53.051714Z","iopub.execute_input":"2022-07-25T23:56:53.052359Z","iopub.status.idle":"2022-07-25T23:56:53.140562Z","shell.execute_reply.started":"2022-07-25T23:56:53.052317Z","shell.execute_reply":"2022-07-25T23:56:53.139610Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Average Price Percentage Change Over the Years","metadata":{}},{"cell_type":"code","source":"df_year_avg_price =pd.DataFrame( df_train_clean.groupby(\"YearBuilt\")[\"SalePrice\"].mean())\ndf_year_avg_price[\"pct_change\"] = df_year_avg_price.pct_change() * 100\ndf_year_avg_price=df_year_avg_price.reset_index()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-07-25T23:56:53.141847Z","iopub.execute_input":"2022-07-25T23:56:53.142257Z","iopub.status.idle":"2022-07-25T23:56:53.154529Z","shell.execute_reply.started":"2022-07-25T23:56:53.142227Z","shell.execute_reply":"2022-07-25T23:56:53.153424Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = go.Figure()\n\nfig.add_trace(go.Scatter(x=df_year_avg_price[\"YearBuilt\"], \n                         y=df_year_avg_price[\"pct_change\"],\n                         name=\"spline\", \n                    line_shape='spline',line=dict(color=\"#1E90ff\")))\n\nfig.add_hline(y=np.max(df_year_avg_price[\"pct_change\"]), line_dash=\"dot\", row=3, col=\"all\",\n              annotation_text=\"Largest % Increase: \"+str(np.floor(np.max(df_year_avg_price[\"pct_change\"]))),\n              annotation_position=\"bottom left\")\n\nfig.add_hline(y=np.min(df_year_avg_price[\"pct_change\"]), line_dash=\"dot\", row=3, col=\"all\",\n              annotation_text=\"Largest % Decrease: \"+str(np.floor(np.min(df_year_avg_price[\"pct_change\"]))),\n              annotation_position=\"bottom left\")\n\nfig.add_vline(x=1929, line_width=3, line_dash=\"dash\", line_color=\"red\",annotation_text=\"1929\")\nfig.add_vline(x=2008, line_width=3, line_dash=\"dash\", line_color=\"red\",annotation_text=\"2008\")\n# fig.add_vrect(x0=np.min(df_year_avg_price[\"YearBuilt\"])+1, x1=1899,fillcolor=\"yellow\",opacity=0.1)\nfig.add_vrect(x0=1950, x1=1979,fillcolor=\"pink\",opacity=0.3,annotation_text=\"Relatively Stable Period (1950-1979)\",annotation_position=\"top left\")\n\nfig.update_traces(mode='lines+markers')\nfig.update_xaxes(title=\"Year Built\")\nfig.update_yaxes(title=\"Percentage Change\")\nfig.update_layout(title=\"Average Sale Price Percentage Change per year\")\nfig.update_layout(legend=dict(y=0.5, traceorder='reversed', font_size=16))","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-07-25T23:56:53.155864Z","iopub.execute_input":"2022-07-25T23:56:53.156553Z","iopub.status.idle":"2022-07-25T23:56:53.250357Z","shell.execute_reply.started":"2022-07-25T23:56:53.156515Z","shell.execute_reply":"2022-07-25T23:56:53.249366Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train_clean = df_train_clean.sort_values(by=\"YearBuilt\",ascending=True)\nfig = make_subplots(\n    rows=2, cols=2, subplot_titles=(\"Per Century\", \"Per Quarter Century\", \"Per Half Century\", \"Per Decade\")\n)\n\nfig.add_trace(go.Box(x=df_train_clean[\"century\"], y=df_train_clean[\"SalePrice\"],showlegend=False,boxpoints=\"all\"), row=1, col=1)\nfig.add_trace(go.Box(x=df_train_clean[\"quarter_century\"], y=df_train_clean[\"SalePrice\"],showlegend=False,boxpoints=\"all\"), row=1, col=2)\nfig.add_trace(go.Box(x=df_train_clean[\"half_century\"], y=df_train_clean[\"SalePrice\"],showlegend=False,boxpoints=\"all\"), row=2, col=1)\nfig.add_trace(go.Box(x=df_train_clean[\"decade\"], y=df_train_clean[\"SalePrice\"],showlegend=False,boxpoints=\"all\"), row=2, col=2)\n\nfig.update_xaxes(title_text=\"Century\", row=1, col=1)\nfig.update_xaxes(title_text=\"Quarter Century\",  row=1, col=2)\nfig.update_xaxes(title_text=\"Half Century\",  row=2, col=1)\nfig.update_xaxes(title_text=\"Decade\", row=2, col=2)\n\nfig.update_yaxes(title_text=\"Sale Price\", row=1, col=1)\nfig.update_yaxes(title_text=\"Sale Price\",  row=1, col=2)\nfig.update_yaxes(title_text=\"Sale Price\", row=2, col=1)\nfig.update_yaxes(title_text=\"Sale Price\", row=2, col=2)\n\nfig.update_layout(title_text=\"House Sale Price Distributions per timeframe\", height=700)\nfig.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-07-25T23:56:53.251709Z","iopub.execute_input":"2022-07-25T23:56:53.252039Z","iopub.status.idle":"2022-07-25T23:56:53.379154Z","shell.execute_reply.started":"2022-07-25T23:56:53.252010Z","shell.execute_reply":"2022-07-25T23:56:53.378010Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Examining Sale Price\n\nHere, we take a look at the Sale Price and some of it's correlated features along with some categorical features. ","metadata":{}},{"cell_type":"markdown","source":"### Examining Sale Price & Ground Living Area\n\nAfter overall quality, Ground Living Area was the next highest coreelated feature with Sale Price. \n\nWe see this graphically, as the area increases, so does the price. We can also see that houses with greater ground living area tend to be of better quality (or is it better quality houses have more ground living area) and higher priced as compared to houses of lower quality.","metadata":{}},{"cell_type":"code","source":"df_train_clean = df_train_clean.sort_values(by=\"OverallQual\",ascending=True)\n# fig = px.scatter(df_train_clean,\n#                  x=\"GrLivArea\",\n#                  y=\"SalePrice\",\n#                  color=\"qual_metric\",\n#                  opacity = .3,\n#                  color_discrete_sequence=[\"darkred\",\"goldenrod\",\"green\",\"dodgerblue\"],\n#                  trendline=\"ols\",\n#                 labels={\"qual_metric\": \"Quality\"})\n\n# fig.update_layout(title=\"Sale Price vs Ground Living Area\")\n# fig.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-07-25T23:56:53.380394Z","iopub.execute_input":"2022-07-25T23:56:53.380697Z","iopub.status.idle":"2022-07-25T23:56:53.389405Z","shell.execute_reply.started":"2022-07-25T23:56:53.380670Z","shell.execute_reply":"2022-07-25T23:56:53.388244Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"From the documentation: \n> This kind of visualization (and the related 2D histogram, or density heatmap) is often used to manage over-plotting, or situations where showing large data sets as scatter plots would result in points overlapping each other and hiding patterns.","metadata":{}},{"cell_type":"code","source":"fig = px.density_contour(df_train_clean,\n                 x=\"GrLivArea\",\n                 y=\"SalePrice\",\n                 color = \"qual_metric\",\n                color_discrete_sequence=[\"darkred\",\"goldenrod\",\"green\",\"dodgerblue\"],labels={\"qual_metric\": \"Quality\"})\nfig.update_traces( contours_showlabels = True)\nfig.update_layout(title=\"Sale Price vs Ground Living Area\")\nfig.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-07-25T23:56:53.391041Z","iopub.execute_input":"2022-07-25T23:56:53.391439Z","iopub.status.idle":"2022-07-25T23:56:53.490123Z","shell.execute_reply.started":"2022-07-25T23:56:53.391401Z","shell.execute_reply":"2022-07-25T23:56:53.488883Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Here we also take a look at Sale Price & Ground Living area, but we also throw in Total Rooms Above Ground into the mix, we also segment this graphically bases on quality. It's really no surprise that there are more rooms above ground as the ground living area increases. (See violin plot/s below).","metadata":{}},{"cell_type":"code","source":"fig = px.scatter(df_train_clean,x=\"GrLivArea\",\n                 y=\"SalePrice\",\n                 facet_col=\"qual_metric\",\n                 color=\"TotRmsAbvGrd\",\n                 opacity = 0.99,\n                 color_continuous_scale=\"teal\",\n                 trendline=\"ols\")\n\nfig.update_layout(title=\"Sale Price vs Ground Living Area\")\nfig.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-07-25T23:56:53.491657Z","iopub.execute_input":"2022-07-25T23:56:53.492013Z","iopub.status.idle":"2022-07-25T23:56:54.288972Z","shell.execute_reply.started":"2022-07-25T23:56:53.491980Z","shell.execute_reply":"2022-07-25T23:56:54.287942Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"As stated earlier, it's no secret that as the quality of the houses increase, so does the ground living area. ","metadata":{}},{"cell_type":"code","source":"fig = px.violin(df_train_clean,y=\"GrLivArea\",\n                color=\"qual_metric\",\n                color_discrete_sequence=[\"darkred\",\"goldenrod\",\"green\",\"dodgerblue\"])\nfig.update_layout(title=\"Distribution of Ground Living Area per House Quality\")\nfig.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-07-25T23:56:54.290414Z","iopub.execute_input":"2022-07-25T23:56:54.290734Z","iopub.status.idle":"2022-07-25T23:56:54.385940Z","shell.execute_reply.started":"2022-07-25T23:56:54.290705Z","shell.execute_reply":"2022-07-25T23:56:54.384870Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Ok, so here the check out the distributions of Total Rooms Above Ground, we see that the median for the \"Worst\" and \"Bad\" quality houses are 6, where as the median is 8 for \"Excellent\" Quality\" Houses. Intuitively, one can think that all these features are interacting with each other. (perhaps a neural network would be a good way to predict Sale price)","metadata":{}},{"cell_type":"code","source":"fig = px.box(df_train_clean,y=\"TotRmsAbvGrd\",\n             color=\"qual_metric\",points=\"outliers\",\n             color_discrete_sequence=[\"darkred\",\"goldenrod\",\"green\",\"dodgerblue\"])\nfig.update_layout(title=\"Distribution of Total Rooms Above Ground per House Quality\")\nfig.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-07-25T23:56:54.387297Z","iopub.execute_input":"2022-07-25T23:56:54.387634Z","iopub.status.idle":"2022-07-25T23:56:54.473467Z","shell.execute_reply.started":"2022-07-25T23:56:54.387598Z","shell.execute_reply":"2022-07-25T23:56:54.472434Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train_clean = df_train_clean.sort_values(by=\"OverallQual\",ascending=True)\n\nfig = px.scatter(df_train_clean,x=\"1stFlrSF\",\n                 y=\"SalePrice\",\n                 facet_col=\"qual_metric\",\n                 opacity=.5,color=\"FullBath\",\n                 color_continuous_scale=\"deep\")\nfig.update_layout(title=\"Sale Price vs 1st Floor Square Footage\")\nfig.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-07-25T23:56:54.474952Z","iopub.execute_input":"2022-07-25T23:56:54.475296Z","iopub.status.idle":"2022-07-25T23:56:54.602800Z","shell.execute_reply.started":"2022-07-25T23:56:54.475266Z","shell.execute_reply":"2022-07-25T23:56:54.601710Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = px.violin(df_train_clean,y=\"1stFlrSF\",\n                color=\"qual_metric\",\n                color_discrete_sequence=[\"darkred\",\"goldenrod\",\"green\",\"dodgerblue\"])\nfig.update_layout(title=\"1st Floor Square Footage Distribution per House Quality\")\n\nfig.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-07-25T23:56:54.604240Z","iopub.execute_input":"2022-07-25T23:56:54.604570Z","iopub.status.idle":"2022-07-25T23:56:54.674034Z","shell.execute_reply.started":"2022-07-25T23:56:54.604541Z","shell.execute_reply":"2022-07-25T23:56:54.672980Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Price Density","metadata":{}},{"cell_type":"markdown","source":"### Price Density Per Neighborhood\n\nHere we see that the \"NAmes\" neighborhood has the most houses valued at 100k - 199k. \n\nNeighborhoods such as NoRidge,NridfHt,SomerSt, StoneBr & Timber have better price diversity as compared to the other neighborhoods. (Check out the parallel coordinates plot to see a differing view of price density.) ","metadata":{}},{"cell_type":"code","source":"fig = px.imshow(pd.crosstab(df_train_clean[\"SalePriceBinned\"],\n                            df_train_clean[\"Neighborhood\"]),\n                color_continuous_scale=\"rdbu\")\nfig.update_layout(title=\"Price Density per Neighborhood\")\nfig.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-07-25T23:56:54.675702Z","iopub.execute_input":"2022-07-25T23:56:54.676581Z","iopub.status.idle":"2022-07-25T23:56:54.750815Z","shell.execute_reply.started":"2022-07-25T23:56:54.676538Z","shell.execute_reply":"2022-07-25T23:56:54.749743Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Almost half of all the houses are 1Fam! & between 100k - 199k !","metadata":{}},{"cell_type":"code","source":"fig = px.imshow(pd.crosstab(df_train_clean[\"SalePriceBinned\"],\n                            df_train_clean[\"BldgType\"]),\n                color_continuous_scale=\"teal\")\nfig.update_layout(title=\"Price Density per Building Type\")\nfig.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-07-25T23:56:54.752365Z","iopub.execute_input":"2022-07-25T23:56:54.752694Z","iopub.status.idle":"2022-07-25T23:56:54.818957Z","shell.execute_reply.started":"2022-07-25T23:56:54.752664Z","shell.execute_reply":"2022-07-25T23:56:54.818231Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = px.imshow(pd.crosstab(df_train_clean[\"SalePriceBinned\"],\n                            df_train_clean[\"decade\"]),\n                color_continuous_scale=\"picnic\")\nfig.update_layout(title=\"Price Density per Decade\")\nfig.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-07-25T23:56:54.820105Z","iopub.execute_input":"2022-07-25T23:56:54.820578Z","iopub.status.idle":"2022-07-25T23:56:54.888259Z","shell.execute_reply.started":"2022-07-25T23:56:54.820549Z","shell.execute_reply":"2022-07-25T23:56:54.887216Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Alternative way to view the price desnity (hover over to see which neighborhood matches to what price).","metadata":{}},{"cell_type":"code","source":"df_train_clean = df_train_clean.sort_values(by=\"OverallQual\",ascending=True)\nfig = px.parallel_categories(df_train_clean,\n                             dimensions=['Neighborhood',\"SalePriceBinned\"],\n                             color=\"YearBuilt\",\n                             color_continuous_scale=px.colors.sequential.Plotly3)\nfig.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-07-25T23:56:54.889570Z","iopub.execute_input":"2022-07-25T23:56:54.889895Z","iopub.status.idle":"2022-07-25T23:56:54.991946Z","shell.execute_reply.started":"2022-07-25T23:56:54.889864Z","shell.execute_reply":"2022-07-25T23:56:54.990830Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = px.parallel_coordinates(df_train_clean, dimensions =[\"GarageArea\",\n                                                           \"GarageCars\",\n                                                           \"GrLivArea\",\n                                                           \"TotRmsAbvGrd\",\n                                                           \"KitchenAbvGr\",\n                                                           \"FullBath\",\n                                                           \"SalePrice\"],\n                             labels = {\"GrLivArea\": \"Living Area\",\"GarageCars\": \"Garage Capacity\",\n                                       \"KitchenAbvGr\": \"# Kit'en abv Grd Flr\",\n                                       \"FullBath\": \"# full baths\",\n                                       \"TotRmsAbvGrd\": \"# Rooms Above Ground\", \n                                       \"GarageArea\": \"Garage Area\",\n                                       \"SalePrice\":\"Sale Price\"},\n                             color_continuous_scale=\"agsunset\",color=\"LotArea\")\nfig.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-07-25T23:56:54.993296Z","iopub.execute_input":"2022-07-25T23:56:54.993635Z","iopub.status.idle":"2022-07-25T23:56:55.079412Z","shell.execute_reply.started":"2022-07-25T23:56:54.993604Z","shell.execute_reply":"2022-07-25T23:56:55.078534Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"From a bird's eye view, we see that most houses, are have a Paved Street, are of building type \"1Fam\", have 1 - 2 Full baths, rank of \"Good\" quality and are in mostly the 100k - 199k Price Range.","metadata":{}},{"cell_type":"code","source":"fig = px.parallel_categories(df_train_clean, dimensions=['Neighborhood',\n                                                         \"Street\",\n                                                         \"BldgType\" ,\n                                                         \"FullBath\",\n                                                         \"qual_metric\",\n                                                         \"SalePriceBinned\"],\n                             color=\"YearBuilt\", \n                             color_continuous_scale=px.colors.sequential.Jet)\nfig.update_layout(title=\"Neighborhood -> Street -> Bldg Type -> # Full Baths -> Quality Rating -> Sale Price\")\nfig.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-07-25T23:56:55.081044Z","iopub.execute_input":"2022-07-25T23:56:55.081708Z","iopub.status.idle":"2022-07-25T23:56:55.200484Z","shell.execute_reply.started":"2022-07-25T23:56:55.081673Z","shell.execute_reply":"2022-07-25T23:56:55.199200Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Drop created columns","metadata":{}},{"cell_type":"code","source":"X_train_temp = df_train_clean.drop([\"Id\",\"SalePrice\",\"qual_metric\",\"SalePriceBinned\",\"decade\",\"quarter_century\",\"half_century\",\"century\"],axis=1)\ny_train = df_train_clean[\"SalePrice\"]\n\ntest_id_cols =list(df_test_clean[\"Id\"])\nX_test_temp = df_test_clean.drop([\"Id\"],axis = 1)","metadata":{"execution":{"iopub.status.busy":"2022-07-25T23:56:55.201984Z","iopub.execute_input":"2022-07-25T23:56:55.203006Z","iopub.status.idle":"2022-07-25T23:56:55.213572Z","shell.execute_reply.started":"2022-07-25T23:56:55.202960Z","shell.execute_reply":"2022-07-25T23:56:55.212449Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Dummy variables for categorical data","metadata":{}},{"cell_type":"code","source":"temp_concat = pd.concat([X_train_temp,X_test_temp],axis = 0)\ntemp_concat = pd.get_dummies(temp_concat)","metadata":{"execution":{"iopub.status.busy":"2022-07-25T23:56:55.215379Z","iopub.execute_input":"2022-07-25T23:56:55.215926Z","iopub.status.idle":"2022-07-25T23:56:55.299050Z","shell.execute_reply.started":"2022-07-25T23:56:55.215893Z","shell.execute_reply":"2022-07-25T23:56:55.298041Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train = temp_concat.iloc[0:1460,:]\nX_test = temp_concat.iloc[1460:,:]","metadata":{"execution":{"iopub.status.busy":"2022-07-25T23:56:55.300303Z","iopub.execute_input":"2022-07-25T23:56:55.300671Z","iopub.status.idle":"2022-07-25T23:56:55.306821Z","shell.execute_reply.started":"2022-07-25T23:56:55.300630Z","shell.execute_reply":"2022-07-25T23:56:55.305727Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(X_train.shape[1]==X_test.shape[1])","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"execution":{"iopub.status.busy":"2022-07-25T23:56:55.308298Z","iopub.execute_input":"2022-07-25T23:56:55.309033Z","iopub.status.idle":"2022-07-25T23:56:55.318930Z","shell.execute_reply.started":"2022-07-25T23:56:55.308991Z","shell.execute_reply":"2022-07-25T23:56:55.318088Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(X_train.shape)\nprint(X_test.shape)\nprint(set(X_train.columns)-set(X_test.columns))","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"execution":{"iopub.status.busy":"2022-07-25T23:56:55.320229Z","iopub.execute_input":"2022-07-25T23:56:55.322462Z","iopub.status.idle":"2022-07-25T23:56:55.331289Z","shell.execute_reply.started":"2022-07-25T23:56:55.322420Z","shell.execute_reply":"2022-07-25T23:56:55.330448Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Feature Scaling\nMost ML algorithms (with the notable exception of tree based estimators) expect/assume the numerical attributes to have similar scales, having different scales causes performance to drop. \n### Original data\nNo scaling is done to the data, the data is left as is in it's original form.","metadata":{}},{"cell_type":"markdown","source":"### StandardScaler\nThe StandardScaler subtracts the mean (standardized values have zero mean) and divides it by the unit variance. Therefore the resulting distribution has unit variance. It should be noted that outliers affects feature standardization using this method, the mean is influenced by outliers and thus each feature may not always have similar scales. Therefore StandardScaler cannot always gurantee balanced feature scales when outliers are in the data. ","metadata":{}},{"cell_type":"code","source":"def scaled_df(X_scaled,y_target,list_of_cols):\n    scaled_df = pd.DataFrame(X_scaled,columns=list_of_cols)\n    scaled_df = pd.concat([scaled_df,y_target],axis=1)\n    return scaled_df","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-07-25T23:56:55.332780Z","iopub.execute_input":"2022-07-25T23:56:55.333300Z","iopub.status.idle":"2022-07-25T23:56:55.344530Z","shell.execute_reply.started":"2022-07-25T23:56:55.333269Z","shell.execute_reply":"2022-07-25T23:56:55.343257Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def scale_comparision(scaled_df,orig_df,x_var,y_var,x_title,y_title,scaler_name):\n    fig = make_subplots(\n        rows=1, cols=2,\n        subplot_titles=(\"Without Scaling\", \"With \"+str(scaler_name)))\n\n    fig.add_trace(go.Scattergl(x=orig_df[x_var],\n                             y=orig_df[y_var],\n                            mode=\"markers\"),\n                  row=1, col=1)\n\n    fig.add_trace(go.Scattergl(x=scaled_df[x_var],\n                             y=scaled_df[y_var],\n                            mode=\"markers\"),\n                  row=1, col=2)\n\n\n\n    fig.update_layout(title_text=\"Comparision of Unscaled and Scaled Data \")\n    fig.update_xaxes(title=x_title)\n    fig.update_yaxes(title=y_title)\n    fig.update_layout(showlegend=False)\n    fig.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-07-25T23:56:55.346044Z","iopub.execute_input":"2022-07-25T23:56:55.347031Z","iopub.status.idle":"2022-07-25T23:56:55.358562Z","shell.execute_reply.started":"2022-07-25T23:56:55.346990Z","shell.execute_reply":"2022-07-25T23:56:55.357673Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import StandardScaler\n\nstd_scaler = StandardScaler()\nX_train_std_scaled = std_scaler.fit_transform(X_train)\nX_train_std_scaled.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-25T23:56:55.359984Z","iopub.execute_input":"2022-07-25T23:56:55.361038Z","iopub.status.idle":"2022-07-25T23:56:55.391545Z","shell.execute_reply.started":"2022-07-25T23:56:55.360997Z","shell.execute_reply":"2022-07-25T23:56:55.390074Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# scale_comparision(scaled_df(X_train_std_scaled,y_train,list(X_train.columns)),\n#                  df_train_clean,\n#                  \"LotArea\",\n#                  \"GrLivArea\",\n#                  \"Lot Area\",\n#                  \"Ground Living Area\",\n#                  \"StandardScaler\")","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-07-25T23:56:55.393649Z","iopub.execute_input":"2022-07-25T23:56:55.394496Z","iopub.status.idle":"2022-07-25T23:56:55.398989Z","shell.execute_reply.started":"2022-07-25T23:56:55.394450Z","shell.execute_reply":"2022-07-25T23:56:55.398243Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### MinMaxScaler\nThis scaling method shifts and rescales the data so it's range is $[0,1]$.\nThis is done by subtracting the minimum value from the feature and then dividing it by the max value of the feature minus the minmum value of the feature. It is then standardized by multiplying it by the $(max-min)+min$ where $max$ and $min$ are the maximum and minimum values of the feature range. This method of scaling, like StandardScaler is very sensitive to outliers.","metadata":{}},{"cell_type":"code","source":"from sklearn.preprocessing import MinMaxScaler\n\nmin_max_scaler = MinMaxScaler()\nX_train_min_max_scaled = min_max_scaler.fit_transform(X_train)\nX_train_min_max_scaled.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-25T23:56:55.400252Z","iopub.execute_input":"2022-07-25T23:56:55.401345Z","iopub.status.idle":"2022-07-25T23:56:55.427998Z","shell.execute_reply.started":"2022-07-25T23:56:55.401295Z","shell.execute_reply":"2022-07-25T23:56:55.426929Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# scale_comparision(scaled_df(X_train_min_max_scaled,y_train,list(X_train.columns)),\n#                  df_train_clean,\n#                  \"LotArea\",\n#                  \"GrLivArea\",\n#                  \"Lot Area\",\n#                  \"Ground Living Area\",\n#                  \"MinMaxScaler\")","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-07-25T23:56:55.429677Z","iopub.execute_input":"2022-07-25T23:56:55.430092Z","iopub.status.idle":"2022-07-25T23:56:55.436719Z","shell.execute_reply.started":"2022-07-25T23:56:55.430053Z","shell.execute_reply":"2022-07-25T23:56:55.435637Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### MaxAbsScaler\nThis scaling method behaves in a similar way to MinMaxScaler. In the case where:\n* Only Positive values are in the feature, the range is $[0,1]$, \n* Only Negative values, the range is from $[-1,0]$ \n* Both negative and positive values are present, the range is from $[-1,1]$. \n\nAs with the previous scalers, this method is also sensitive to outliers. ","metadata":{}},{"cell_type":"code","source":"from sklearn.preprocessing import MaxAbsScaler\n\nmax_abs_scaler = MaxAbsScaler()\nX_train_max_abs_scaled = max_abs_scaler.fit_transform(X_train)\nX_train_max_abs_scaled.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-25T23:56:55.438563Z","iopub.execute_input":"2022-07-25T23:56:55.439308Z","iopub.status.idle":"2022-07-25T23:56:55.471781Z","shell.execute_reply.started":"2022-07-25T23:56:55.439266Z","shell.execute_reply":"2022-07-25T23:56:55.470612Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"scale_comparision(scaled_df(X_train_max_abs_scaled,y_train,list(X_train.columns)),\n                 df_train_clean,\n                 \"LotArea\",\n                 \"GrLivArea\",\n                 \"Lot Area\",\n                 \"Ground Living Area\",\n                 \"MaxAbsScaler\")","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-07-25T23:56:55.473034Z","iopub.execute_input":"2022-07-25T23:56:55.473862Z","iopub.status.idle":"2022-07-25T23:56:55.524586Z","shell.execute_reply.started":"2022-07-25T23:56:55.473817Z","shell.execute_reply":"2022-07-25T23:56:55.523481Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### RobustScaler\nThis scaling method scales the data by removing the median and scaling accoriding to the quantile range where the default is the IQR (interquartile range, between the 25th and 75th quantile). RobustScaler is therefore not influenced by outliers. The transformed feature value ranges are larger compared to the MinMax, MaxAbs and Standard scalers. The transformed features are on approximately similar scales as well. However, it should be noted that the outliers themselves are present in the transformed data. ","metadata":{}},{"cell_type":"code","source":"from sklearn.preprocessing import RobustScaler\n\nrobust_scaler = RobustScaler()\nX_train_robust_scaled = robust_scaler.fit_transform(X_train)\nX_train_robust_scaled.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-25T23:56:55.525978Z","iopub.execute_input":"2022-07-25T23:56:55.526955Z","iopub.status.idle":"2022-07-25T23:56:55.599285Z","shell.execute_reply.started":"2022-07-25T23:56:55.526905Z","shell.execute_reply":"2022-07-25T23:56:55.598455Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"scale_comparision(scaled_df(X_train_robust_scaled,y_train,list(X_train.columns)),\n                 df_train_clean,\n                 \"LotArea\",\n                 \"GrLivArea\",\n                 \"Lot Area\",\n                 \"Ground Living Area\",\n                 \"RobustScaler\")","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-07-25T23:56:55.600857Z","iopub.execute_input":"2022-07-25T23:56:55.601343Z","iopub.status.idle":"2022-07-25T23:56:55.656203Z","shell.execute_reply.started":"2022-07-25T23:56:55.601298Z","shell.execute_reply":"2022-07-25T23:56:55.655380Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### PowerTransformer\nA power transformation is applied to each feature to make the data have a gaussian/normal like distribution, this is done to stabilize variance and to minimize skewness (Skewness is the measure of asymmetery in a distribution).","metadata":{}},{"cell_type":"code","source":"from sklearn.preprocessing import PowerTransformer\n\n# the box-cox method can only be applied strictly to positive data\npow_transformer = PowerTransformer(method = \"yeo-johnson\")\nX_train_pow_transform = pow_transformer.fit_transform(X_train)\nX_train_pow_transform.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-25T23:56:55.657380Z","iopub.execute_input":"2022-07-25T23:56:55.658047Z","iopub.status.idle":"2022-07-25T23:56:56.554509Z","shell.execute_reply.started":"2022-07-25T23:56:55.658014Z","shell.execute_reply":"2022-07-25T23:56:56.553395Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# scale_comparision(scaled_df(X_train_pow_transform,y_train,list(X_train.columns)),\n#                  df_train_clean,\n#                  \"LotArea\",\n#                  \"GrLivArea\",\n#                  \"Lot Area\",\n#                  \"Ground Living Area\",\n#                  \"PowerTransformer\")","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-07-25T23:56:56.556007Z","iopub.execute_input":"2022-07-25T23:56:56.557100Z","iopub.status.idle":"2022-07-25T23:56:56.562158Z","shell.execute_reply.started":"2022-07-25T23:56:56.557055Z","shell.execute_reply":"2022-07-25T23:56:56.561187Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### QuantileTransformer (Uniform output/Gaussian output)\nA non linear transfromation is applied s.t the PDF (probabilty density function) for each feature will be mapped to either a Uniform or Gaussian (Normal) distribution. This method of transformation, like RobustScaler are not sensitive to outliers, however, QuantileTransformer will collapse the outliers into the defined range. The transformation is done on each feature independently. ","metadata":{}},{"cell_type":"code","source":"from sklearn.preprocessing import QuantileTransformer","metadata":{"execution":{"iopub.status.busy":"2022-07-25T23:56:56.563626Z","iopub.execute_input":"2022-07-25T23:56:56.564614Z","iopub.status.idle":"2022-07-25T23:56:56.574054Z","shell.execute_reply.started":"2022-07-25T23:56:56.564569Z","shell.execute_reply":"2022-07-25T23:56:56.572706Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"qt_transformer = QuantileTransformer(output_distribution = \"uniform\",\n                                     random_state = RANDOM_STATE)\nX_qt_transform_uni = qt_transformer.fit_transform(X_train)","metadata":{"execution":{"iopub.status.busy":"2022-07-25T23:56:56.575732Z","iopub.execute_input":"2022-07-25T23:56:56.576885Z","iopub.status.idle":"2022-07-25T23:56:57.343516Z","shell.execute_reply.started":"2022-07-25T23:56:56.576844Z","shell.execute_reply":"2022-07-25T23:56:57.342275Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# scale_comparision(scaled_df(X_qt_transform_uni,y_train,list(X_train.columns)),\n#                  df_train_clean,\n#                  \"LotArea\",\n#                  \"GrLivArea\",\n#                  \"Lot Area\",\n#                  \"Ground Living Area\",\n#                  \"QuantileTransformer (Uniform)\")","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-07-25T23:56:57.345169Z","iopub.execute_input":"2022-07-25T23:56:57.345730Z","iopub.status.idle":"2022-07-25T23:56:57.350201Z","shell.execute_reply.started":"2022-07-25T23:56:57.345696Z","shell.execute_reply":"2022-07-25T23:56:57.349249Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"qt_transformer = QuantileTransformer(output_distribution = \"normal\",\n                                     random_state = RANDOM_STATE)\nX_qt_transform_normal = qt_transformer.fit_transform(X_train)\nX_qt_transform_normal.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-25T23:56:57.351595Z","iopub.execute_input":"2022-07-25T23:56:57.352610Z","iopub.status.idle":"2022-07-25T23:56:58.135143Z","shell.execute_reply.started":"2022-07-25T23:56:57.352577Z","shell.execute_reply":"2022-07-25T23:56:58.133834Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"scale_comparision(scaled_df(X_qt_transform_normal,y_train,list(X_train.columns)),\n                 df_train_clean,\n                 \"LotArea\",\n                 \"GrLivArea\",\n                 \"Lot Area\",\n                 \"Ground Living Area\",\n                 \"QuantileTransformer(Normal)\")","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-07-25T23:56:58.136534Z","iopub.execute_input":"2022-07-25T23:56:58.136926Z","iopub.status.idle":"2022-07-25T23:56:58.187517Z","shell.execute_reply.started":"2022-07-25T23:56:58.136893Z","shell.execute_reply":"2022-07-25T23:56:58.186465Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Normalizer\nUnlike the previous scalers where features (columns) are scaled, for normalization the samples (rows) are scaled to have unit norm indpendent of the distribution of the samples. \n\nTypes of unit norm:\n* $l^1$ $normalization$: Summing the absolute values of the normalized vector is equal to 1. \n* $l^2$ $normalization$: If each element in the normalized vector is squared and then summed,it would be equal to 1. This is the most commonly used vector normalization method\n* $l^{\\infty}$ $normalization$: also called Vector Max Norm. Gives the largest magnitude among each element of a vector","metadata":{}},{"cell_type":"code","source":"from sklearn.preprocessing import Normalizer","metadata":{"execution":{"iopub.status.busy":"2022-07-25T23:56:58.189060Z","iopub.execute_input":"2022-07-25T23:56:58.189729Z","iopub.status.idle":"2022-07-25T23:56:58.194935Z","shell.execute_reply.started":"2022-07-25T23:56:58.189689Z","shell.execute_reply":"2022-07-25T23:56:58.193636Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"normalizer_l1 = Normalizer(norm = \"l1\")\nX_train_normalize_l1 = normalizer_l1.fit_transform(X_train)\nX_train_normalize_l1.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-25T23:56:58.196305Z","iopub.execute_input":"2022-07-25T23:56:58.196948Z","iopub.status.idle":"2022-07-25T23:56:58.226217Z","shell.execute_reply.started":"2022-07-25T23:56:58.196912Z","shell.execute_reply":"2022-07-25T23:56:58.224985Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# scale_comparision(scaled_df(X_train_normalize_l1,y_train,list(X_train.columns)),\n#                  df_train_clean,\n#                  \"LotArea\",\n#                  \"GrLivArea\",\n#                  \"Lot Area\",\n#                  \"Ground Living Area\",\n#                  \"l1 Normalization\")","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-07-25T23:56:58.227744Z","iopub.execute_input":"2022-07-25T23:56:58.228686Z","iopub.status.idle":"2022-07-25T23:56:58.233207Z","shell.execute_reply.started":"2022-07-25T23:56:58.228653Z","shell.execute_reply":"2022-07-25T23:56:58.232140Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"normalizer_l2 = Normalizer(norm = \"l2\")\nX_train_normalize_l2 = normalizer_l2.fit_transform(X_train)\nX_train_normalize_l2.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-25T23:56:58.234876Z","iopub.execute_input":"2022-07-25T23:56:58.235665Z","iopub.status.idle":"2022-07-25T23:56:58.256187Z","shell.execute_reply.started":"2022-07-25T23:56:58.235622Z","shell.execute_reply":"2022-07-25T23:56:58.254988Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"scale_comparision(scaled_df(X_train_normalize_l2,y_train,list(X_train.columns)),\n                 df_train_clean,\n                 \"LotArea\",\n                 \"GrLivArea\",\n                 \"Lot Area\",\n                 \"Ground Living Area\",\n                 \"l2 Normalization\")","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-07-25T23:56:58.259075Z","iopub.execute_input":"2022-07-25T23:56:58.259449Z","iopub.status.idle":"2022-07-25T23:56:58.308279Z","shell.execute_reply.started":"2022-07-25T23:56:58.259417Z","shell.execute_reply":"2022-07-25T23:56:58.307422Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"normalizer_max = Normalizer(norm = \"max\")\nX_train_normalize_max = normalizer_max.fit_transform(X_train)\nX_train_normalize_max.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-25T23:56:58.309756Z","iopub.execute_input":"2022-07-25T23:56:58.310103Z","iopub.status.idle":"2022-07-25T23:56:58.333114Z","shell.execute_reply.started":"2022-07-25T23:56:58.310072Z","shell.execute_reply":"2022-07-25T23:56:58.332100Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# scale_comparision(scaled_df(X_train_normalize_max,y_train,list(X_train.columns)),\n#                  df_train_clean,\n#                  \"LotArea\",\n#                  \"GrLivArea\",\n#                  \"Lot Area\",\n#                  \"Ground Living Area\",\n#                  \"Max Normalization\")","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-07-25T23:56:58.334628Z","iopub.execute_input":"2022-07-25T23:56:58.335277Z","iopub.status.idle":"2022-07-25T23:56:58.339567Z","shell.execute_reply.started":"2022-07-25T23:56:58.335242Z","shell.execute_reply":"2022-07-25T23:56:58.338573Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Dimenonsality Reduction","metadata":{}},{"cell_type":"markdown","source":"## Principal Component Analysis (PCA)\n\nThis is perhaps the most widely used technique for dimensionality reduction. The goal of PCA is to project the data onto a space having a dimensionalty of $M$ $<$ $D$, where $D$ is the number of dimensions in the given data set whilst simultaneously maximizing the variance of the projected data. The principal componenets are founding using *Singular Value Decomposition (SVD)*, the training set is decomposed into the multiplication of three matricies, say $UEV^T$, where $V$ contains the unit vectors that have all the principal components of interest. \n\nA point of interest is the *explained variance ratio* of each principal component. In a nutshell, *explained variance ratio* is the proportion of the data's variance along each principal component. (Refer to graphs for a visual understanding)\n\n\n","metadata":{}},{"cell_type":"code","source":"from sklearn.decomposition import PCA","metadata":{"execution":{"iopub.status.busy":"2022-07-25T23:56:58.356376Z","iopub.execute_input":"2022-07-25T23:56:58.357386Z","iopub.status.idle":"2022-07-25T23:56:58.361633Z","shell.execute_reply.started":"2022-07-25T23:56:58.357345Z","shell.execute_reply":"2022-07-25T23:56:58.360477Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def disp_explained_variance_ratio(X,title):\n    \n    pca = PCA(n_components=0.9999)\n    pca.fit(X)\n    exp_var_cumul = np.cumsum(pca.explained_variance_ratio_)\n    \n    explained_variance = px.area(\n        x=range(1, exp_var_cumul.shape[0] + 1),\n        y=exp_var_cumul,\n        color_discrete_sequence=[\"deepskyblue\"],\n        labels={\"x\": \"# Components\", \"y\": \"Explained Variance\"}\n    )\n    \n    explained_variance.update_layout(title=title)\n    explained_variance.show()\n    \n    dimensions = px.bar(x=range(pca.n_components_), y=pca.explained_variance_ratio_,\n                        color_discrete_sequence=[\"deeppink\"],\n                        labels={\"x\":\"Component #\",\"y\":\"Explained Variance\"})\n    dimensions.show()\n    ","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-07-25T23:56:58.362876Z","iopub.execute_input":"2022-07-25T23:56:58.363224Z","iopub.status.idle":"2022-07-25T23:56:58.374642Z","shell.execute_reply.started":"2022-07-25T23:56:58.363187Z","shell.execute_reply":"2022-07-25T23:56:58.373848Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"disp_explained_variance_ratio(X_train,title=\"PCA 99.99% explained variance using Unscaled data\")","metadata":{"execution":{"iopub.status.busy":"2022-07-25T23:56:58.375902Z","iopub.execute_input":"2022-07-25T23:56:58.376768Z","iopub.status.idle":"2022-07-25T23:56:58.668789Z","shell.execute_reply.started":"2022-07-25T23:56:58.376737Z","shell.execute_reply":"2022-07-25T23:56:58.667753Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# disp_explained_variance_ratio(X_train_std_scaled,title=\"PCA using Standard Scaler data\")","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"execution":{"iopub.status.busy":"2022-07-25T23:56:58.670263Z","iopub.execute_input":"2022-07-25T23:56:58.670753Z","iopub.status.idle":"2022-07-25T23:56:58.675056Z","shell.execute_reply.started":"2022-07-25T23:56:58.670720Z","shell.execute_reply":"2022-07-25T23:56:58.674106Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# disp_explained_variance_ratio(X_train_min_max_scaled,title=\"PCA using MinMax Scaler data\")","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"execution":{"iopub.status.busy":"2022-07-25T23:56:58.676347Z","iopub.execute_input":"2022-07-25T23:56:58.677043Z","iopub.status.idle":"2022-07-25T23:56:58.686814Z","shell.execute_reply.started":"2022-07-25T23:56:58.677013Z","shell.execute_reply":"2022-07-25T23:56:58.685728Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# disp_explained_variance_ratio(X_train_max_abs_scaled,title=\"PCA using MaxAbs data\")","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"execution":{"iopub.status.busy":"2022-07-25T23:56:58.688459Z","iopub.execute_input":"2022-07-25T23:56:58.689138Z","iopub.status.idle":"2022-07-25T23:56:58.697863Z","shell.execute_reply.started":"2022-07-25T23:56:58.689095Z","shell.execute_reply":"2022-07-25T23:56:58.696833Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"disp_explained_variance_ratio(X_train_robust_scaled,title=\"PCA 99.99% explained variance using Robust Scaler data\")","metadata":{"execution":{"iopub.status.busy":"2022-07-25T23:56:58.699293Z","iopub.execute_input":"2022-07-25T23:56:58.700248Z","iopub.status.idle":"2022-07-25T23:56:58.965730Z","shell.execute_reply.started":"2022-07-25T23:56:58.700205Z","shell.execute_reply":"2022-07-25T23:56:58.964613Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# disp_explained_variance_ratio(X_train_pow_transform,\n#                               title=\"PCA explained variance using Power Transformer data\")","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"execution":{"iopub.status.busy":"2022-07-25T23:56:58.967383Z","iopub.execute_input":"2022-07-25T23:56:58.968053Z","iopub.status.idle":"2022-07-25T23:56:58.972917Z","shell.execute_reply.started":"2022-07-25T23:56:58.968010Z","shell.execute_reply":"2022-07-25T23:56:58.971685Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# disp_explained_variance_ratio(X_qt_transform_uni,\n#                               title=\"PCA explained variance using Quantile Transformer (uniform) data\")","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"execution":{"iopub.status.busy":"2022-07-25T23:56:58.974412Z","iopub.execute_input":"2022-07-25T23:56:58.975337Z","iopub.status.idle":"2022-07-25T23:56:58.983051Z","shell.execute_reply.started":"2022-07-25T23:56:58.975295Z","shell.execute_reply":"2022-07-25T23:56:58.982329Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"disp_explained_variance_ratio(X_qt_transform_normal,\n                              title=\"PCA explained variance using Quantile Transformer (normal) data\")","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"execution":{"iopub.status.busy":"2022-07-25T23:56:58.984495Z","iopub.execute_input":"2022-07-25T23:56:58.985247Z","iopub.status.idle":"2022-07-25T23:56:59.255310Z","shell.execute_reply.started":"2022-07-25T23:56:58.985205Z","shell.execute_reply":"2022-07-25T23:56:59.254251Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# disp_explained_variance_ratio(X_train_normalize_l1,\n#                               title=\"PCA 99.99% explained variance using l1 data\")","metadata":{"execution":{"iopub.status.busy":"2022-07-25T23:56:59.256693Z","iopub.execute_input":"2022-07-25T23:56:59.257770Z","iopub.status.idle":"2022-07-25T23:56:59.262668Z","shell.execute_reply.started":"2022-07-25T23:56:59.257728Z","shell.execute_reply":"2022-07-25T23:56:59.261248Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"disp_explained_variance_ratio(X_train_normalize_l2,\n                              title=\"PCA explained variance using l2 data\")\n","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"execution":{"iopub.status.busy":"2022-07-25T23:56:59.264300Z","iopub.execute_input":"2022-07-25T23:56:59.265053Z","iopub.status.idle":"2022-07-25T23:56:59.538256Z","shell.execute_reply.started":"2022-07-25T23:56:59.265012Z","shell.execute_reply":"2022-07-25T23:56:59.537500Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# disp_explained_variance_ratio(X_train_normalize_max,\n#                               title=\"PCA explained variance using Max Normalization data\")","metadata":{"_kg_hide-output":true,"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-07-25T23:56:59.539475Z","iopub.execute_input":"2022-07-25T23:56:59.539965Z","iopub.status.idle":"2022-07-25T23:56:59.543914Z","shell.execute_reply.started":"2022-07-25T23:56:59.539932Z","shell.execute_reply":"2022-07-25T23:56:59.543152Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Variance and Number of Componenents\nWe see that PCA retained a 99.99% variance with 8 features out of 270 features. This is a 97% reduction in dimensionality white maintaining a very high amount of variance. If we were to tolerate a threshold of 95% variance, we would only need 2 out of the 270 components, a 99.2% reduction in dimensionality!\n\nBy reducing the dimensions, we are signficantly able to speed up training time.","metadata":{}},{"cell_type":"markdown","source":"## Using Feature Importance\n\n### Using a Random Forest","metadata":{}},{"cell_type":"code","source":"from sklearn.ensemble import RandomForestRegressor\n\nrnd_forest_reg = RandomForestRegressor(n_estimators=1500,max_depth=1,max_features=\"sqrt\",criterion=\"squared_error\")\nrnd_forest_reg.fit(X_train_robust_scaled,y_train)","metadata":{"execution":{"iopub.status.busy":"2022-07-25T23:56:59.545041Z","iopub.execute_input":"2022-07-25T23:56:59.546017Z","iopub.status.idle":"2022-07-25T23:57:01.624942Z","shell.execute_reply.started":"2022-07-25T23:56:59.545985Z","shell.execute_reply":"2022-07-25T23:57:01.623867Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"features = sorted(list(zip(X_train.columns,rnd_forest_reg.feature_importances_)),key=lambda x: x[1],reverse=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-25T23:57:01.626155Z","iopub.execute_input":"2022-07-25T23:57:01.626529Z","iopub.status.idle":"2022-07-25T23:57:01.765313Z","shell.execute_reply.started":"2022-07-25T23:57:01.626496Z","shell.execute_reply":"2022-07-25T23:57:01.764219Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Let's check out the top 20 features","metadata":{}},{"cell_type":"code","source":"n_features = 20\nx = [features[i][0] for i in range(0,n_features)]\ny = [features[i][1] for i in range(0,n_features)]","metadata":{"execution":{"iopub.status.busy":"2022-07-25T23:57:01.766878Z","iopub.execute_input":"2022-07-25T23:57:01.767246Z","iopub.status.idle":"2022-07-25T23:57:01.772608Z","shell.execute_reply.started":"2022-07-25T23:57:01.767211Z","shell.execute_reply":"2022-07-25T23:57:01.771534Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig= px.bar(x=x,y=y,color_discrete_sequence=[\"deepskyblue\"])\nfig.update_xaxes(title=\"Features\")\nfig.update_yaxes(title=\"Relative Importance\")\nfig.update_layout(title = \"Top 20 Features using Random Forest\")\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-25T23:57:01.773599Z","iopub.execute_input":"2022-07-25T23:57:01.773919Z","iopub.status.idle":"2022-07-25T23:57:01.841599Z","shell.execute_reply.started":"2022-07-25T23:57:01.773890Z","shell.execute_reply":"2022-07-25T23:57:01.840740Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Using Lasso ","metadata":{}},{"cell_type":"code","source":"from sklearn.linear_model import Lasso \n\nlasso = Lasso(alpha=10.0,fit_intercept=True)\nlasso.fit(X_train_robust_scaled,y_train)","metadata":{"execution":{"iopub.status.busy":"2022-07-25T23:57:01.843201Z","iopub.execute_input":"2022-07-25T23:57:01.843509Z","iopub.status.idle":"2022-07-25T23:57:02.290642Z","shell.execute_reply.started":"2022-07-25T23:57:01.843481Z","shell.execute_reply":"2022-07-25T23:57:02.289424Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"features = sorted(list(zip(X_train.columns,np.abs(lasso.coef_))),key=lambda x:x[1],reverse=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-25T23:57:02.292451Z","iopub.execute_input":"2022-07-25T23:57:02.293610Z","iopub.status.idle":"2022-07-25T23:57:02.300644Z","shell.execute_reply.started":"2022-07-25T23:57:02.293559Z","shell.execute_reply":"2022-07-25T23:57:02.299449Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Let's check out the top 20 features","metadata":{}},{"cell_type":"code","source":"n_features = 20\nx = [features[i][0] for i in range(0,n_features)]\ny = [features[i][1] for i in range(0,n_features)]","metadata":{"execution":{"iopub.status.busy":"2022-07-25T23:57:02.302482Z","iopub.execute_input":"2022-07-25T23:57:02.303242Z","iopub.status.idle":"2022-07-25T23:57:02.313035Z","shell.execute_reply.started":"2022-07-25T23:57:02.303194Z","shell.execute_reply":"2022-07-25T23:57:02.311910Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig= px.bar(x=x,y=y,color_discrete_sequence=[\"dodgerblue\"])\nfig.update_xaxes(title=\"Features\")\nfig.update_yaxes(title=\"Coefficent\")\nfig.update_layout(title = \"Top 20 Features using Lasso Regression\")\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-25T23:57:02.314923Z","iopub.execute_input":"2022-07-25T23:57:02.315648Z","iopub.status.idle":"2022-07-25T23:57:02.415028Z","shell.execute_reply.started":"2022-07-25T23:57:02.315606Z","shell.execute_reply":"2022-07-25T23:57:02.414219Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Pipeline","metadata":{}},{"cell_type":"code","source":"from sklearn.pipeline import Pipeline\n# For Faster running\n# pipeline = Pipeline([\n#     (\"rob_scaler\",RobustScaler()),\n#     (\"pca\",PCA(n_components=1,whiten=True))\n# ])\n\npipeline = Pipeline([\n    (\"rob_scaler\",RobustScaler()),\n    (\"pca\",PCA(n_components=31,whiten = True))\n])\n\nX_train_prepped = pipeline.fit_transform(X_train)\nX_test_prepped = pipeline.transform(X_test)","metadata":{"execution":{"iopub.status.busy":"2022-07-25T23:57:02.416340Z","iopub.execute_input":"2022-07-25T23:57:02.416882Z","iopub.status.idle":"2022-07-25T23:57:02.564451Z","shell.execute_reply.started":"2022-07-25T23:57:02.416849Z","shell.execute_reply":"2022-07-25T23:57:02.563071Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#  Modelling \n\nModels being used:\n- Linear Regression (Ordinary Least Squares)\n- Ridge Regression\n- Lasso Regression\n- Elastic Net\n- Gradient Descent\n- Polynomial Regression\n- KNN Regression\n- RandomForest Regression\n- AdaBoostRegressor\n- GradientBoostingRegressor\n- HistGradientBoostingRegressor\n- StackingRegressor","metadata":{}},{"cell_type":"markdown","source":"# Linear [Regression](https://youtu.be/-pBwIsVGaL4)","metadata":{}},{"cell_type":"markdown","source":"## Linear Regression\n\n### Overview\n**Linear Regresion**, or sometimes called **Simple Linear Regression** provides a model to measure the magnitude between one variable and that of a second variable. It is a way of predicting some quantitative response $Y$ based on a *single* predicor $X$. This regression equation can be written as,\n\n$$Y = \\hat{b_0} + \\hat{b_1}X$$\n\n- $\\hat{b_0}$ is known as the intercept \n- $\\hat{b_1}$ is known as the slope for $X$\n- $\\hat{b_0}$ and $\\hat{b_1}$ are known as the coefficents/parameters, it should be noted that in general, the term $b_1$ is called the coefficent\n- $Y$ is the response/dependent variable \n- $X$ is the predictor/independent variable\n\nIn order to measure the coefficents, the most common way this is done is **Ordinary Least Squares**. The coefficents $b_0$ and $b_1$ are usually unknownand as such, we want to find the slope and intercept such that the line is as close as possible to the data points. \n\n The prediction, for say some $ith$ value of $X$ is $\\hat{y_i}$ = $\\hat{b_0}$+$\\hat{b_1}x_i$, then\n $$e_i = y_i - \\hat{y_i}$$\n This is how the residuals are computed, by subtracting the predicted values from the orignal data.\n\nThe residual sum of squares is defined as:\n\n$$RSS = e_1^2+e_2^2+...+e_n^2$$\n\nAnother way of writing this,\n$$e_i = y_i -\\hat{b_0}-\\hat{b_1}x_i$$\n$$RSS =(y_1 -\\hat{b_0}-\\hat{b_1}x_1)^2+(y_2 -\\hat{b_0}-\\hat{b_1}x_2)^2+...+(y_n -\\hat{b_0}-\\hat{b_1}x_n)^2 $$\n\nThe minimizers for $\\hat{b_0}$ and $\\hat{b_1}$ are:\n\n$$\\hat{b_1} = \\frac{\\sum\\limits_{i=1}^{n} (x_i-\\bar{x})(y_i-\\bar{y})}{\\sum\\limits_{i=1}^{n} (x_i-\\bar{x})^2}$$\n\n$$\\hat{b_0}=\\bar{y}-\\hat{b_1}\\bar{x}$$\n\n$\\bar{y}$, $\\bar{x}$ are the sample means\n\n### Assessing the accuracy of the coefficent estimates\n$$Y = {b_0}+{b_1}X_1 +\\epsilon$$\n\n$\\epsilon$ is usually referred to as a random deviation or random error term in the model. Without this term, all the points $(x,y)$ would correspond to a point falling exactly on the line $Y = \\hat{b_0} + \\hat{b_1}X$. If $\\epsilon >0$ this will cause the points to fall above the true regression line or below if $\\epsilon<0$. This error term is assumed to be independent of $X$.\n\nThis model is the population regresson line which is the best linear approximation to the true relationship between $X$ and $Y$. In most practical aspects, we compute the least squares line from our set of observations but the population regression line is unobserved as we generally do not know the true relationship for the real data.\n\nTo find out how close $\\hat{b_0}$ and $\\hat{b_1}$ are to the true values of ${b_0}$ and ${b_1}$, we can compute the standard error associated with $\\hat{b_0}$ and $\\hat{b_1}$\n\n$$SE(\\hat{b_0})^2=\\sigma^2 [\\frac{1}{n}+\\frac{\\bar{x}^2}{\\sum\\limits_{i=1}^{n}(x_i - \\bar{x})^2}]$$\n\n$$SE(\\hat{b_1})^2= [\\frac{\\sigma^2}{\\sum\\limits_{i=1}^{n}(x_i - \\bar{x})^2}]$$\n\n$\\sigma^2=Var(\\epsilon)$, this can be estimated from the data.\n\nWe can use these standard errors to compute confidence intervals and also perform hypothesis tests on these coefficents \n\n## Multiple Linear Regression\n\nIn the real world, there are usually more than one predictors of some target variable. Each predictor is given it's own separate slope coefficent in a single model. Assuming that there are $p$ predictors, then the question takes the form: \n\n$$Y = {b_0}+{b_1}X_1+{b_2}X_2+...++{b_p}X_p+\\epsilon$$\n$X_i$ represents the $ith$ predictor and $b_i$ is the quantifiable association with the variable and response. $b_i$ is interprted as the average effect on Y of a one unit increase in $X_i$ when all the other predictors are held fixed.","metadata":{}},{"cell_type":"code","source":"from sklearn.linear_model import LinearRegression\nfrom sklearn.linear_model import Ridge\nfrom sklearn.linear_model import Lasso \nfrom sklearn.linear_model import ElasticNet\nfrom sklearn.linear_model import SGDRegressor\n\nfrom sklearn.neighbors import KNeighborsRegressor\n\nfrom sklearn.tree import DecisionTreeRegressor\n\nfrom sklearn.ensemble import BaggingRegressor\nfrom sklearn.ensemble import RandomForestRegressor\nfrom sklearn.ensemble import ExtraTreesRegressor\n\nfrom sklearn.ensemble import AdaBoostRegressor\nfrom sklearn.ensemble import GradientBoostingRegressor\nfrom sklearn.ensemble import HistGradientBoostingRegressor\n\nfrom sklearn.ensemble import StackingRegressor\n\nfrom sklearn.model_selection import cross_val_score\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.model_selection import GridSearchCV\nfrom sklearn.model_selection import RandomizedSearchCV\n\nfrom sklearn.preprocessing import PolynomialFeatures\n\nfrom sklearn.metrics import mean_squared_error as MSE\nfrom sklearn.metrics import mean_absolute_error as MAE","metadata":{"execution":{"iopub.status.busy":"2022-07-25T23:57:02.566456Z","iopub.execute_input":"2022-07-25T23:57:02.568393Z","iopub.status.idle":"2022-07-25T23:57:02.590726Z","shell.execute_reply.started":"2022-07-25T23:57:02.568341Z","shell.execute_reply":"2022-07-25T23:57:02.588236Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"best_estimators = set()","metadata":{"execution":{"iopub.status.busy":"2022-07-25T23:57:02.594295Z","iopub.execute_input":"2022-07-25T23:57:02.597333Z","iopub.status.idle":"2022-07-25T23:57:02.608029Z","shell.execute_reply.started":"2022-07-25T23:57:02.597272Z","shell.execute_reply":"2022-07-25T23:57:02.606607Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sk_split = StratifiedKFold(n_splits=3,\n                           random_state=RANDOM_STATE,\n                           shuffle=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-25T23:57:02.614012Z","iopub.execute_input":"2022-07-25T23:57:02.618616Z","iopub.status.idle":"2022-07-25T23:57:02.628836Z","shell.execute_reply.started":"2022-07-25T23:57:02.618537Z","shell.execute_reply":"2022-07-25T23:57:02.627511Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Simple Linear Regression Example\n\nUsing the first floor square footage (independent variable) to predict the Sale Price (dependent variable)","metadata":{}},{"cell_type":"code","source":"from statsmodels.formula.api import ols","metadata":{"execution":{"iopub.status.busy":"2022-07-25T23:57:02.630631Z","iopub.execute_input":"2022-07-25T23:57:02.632688Z","iopub.status.idle":"2022-07-25T23:57:02.647322Z","shell.execute_reply.started":"2022-07-25T23:57:02.632637Z","shell.execute_reply":"2022-07-25T23:57:02.646052Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y = \"SalePrice\"\nx = \"GrLivArea\"\nsimp_lin = df_train_clean[[x,y]]","metadata":{"execution":{"iopub.status.busy":"2022-07-25T23:57:02.648968Z","iopub.execute_input":"2022-07-25T23:57:02.649759Z","iopub.status.idle":"2022-07-25T23:57:02.661545Z","shell.execute_reply.started":"2022-07-25T23:57:02.649708Z","shell.execute_reply":"2022-07-25T23:57:02.660085Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# px.scatter(simp_lin,x=x,y=y,trendline=\"ols\")","metadata":{"execution":{"iopub.status.busy":"2022-07-25T23:57:02.663311Z","iopub.execute_input":"2022-07-25T23:57:02.664093Z","iopub.status.idle":"2022-07-25T23:57:02.673375Z","shell.execute_reply.started":"2022-07-25T23:57:02.664043Z","shell.execute_reply":"2022-07-25T23:57:02.671834Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"simp_lin_model = ols(\"SalePrice ~ GrLivArea\",data=simp_lin).fit()\nprint(simp_lin_model.params)","metadata":{"execution":{"iopub.status.busy":"2022-07-25T23:57:02.675603Z","iopub.execute_input":"2022-07-25T23:57:02.676720Z","iopub.status.idle":"2022-07-25T23:57:02.692434Z","shell.execute_reply.started":"2022-07-25T23:57:02.676670Z","shell.execute_reply":"2022-07-25T23:57:02.691446Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"$$SalePrice = 18569 + (107 * GrLivArea)$$","metadata":{}},{"cell_type":"code","source":"min_gr_liv_area = np.min(simp_lin[\"GrLivArea\"])\nmax_gr_liv_area = np.max(simp_lin[\"GrLivArea\"])\nrange_gr_liv_area = np.arange(min_gr_liv_area,max_gr_liv_area+1,1)\nrange_gr_liv_area\ngr_liv_area_data = pd.DataFrame(range_gr_liv_area,columns=[\"GrLivArea\"])","metadata":{"execution":{"iopub.status.busy":"2022-07-25T23:57:02.693777Z","iopub.execute_input":"2022-07-25T23:57:02.696524Z","iopub.status.idle":"2022-07-25T23:57:02.704143Z","shell.execute_reply.started":"2022-07-25T23:57:02.696478Z","shell.execute_reply":"2022-07-25T23:57:02.703236Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred_data = gr_liv_area_data.assign(predicted_SalePrice=simp_lin_model.predict(gr_liv_area_data))","metadata":{"execution":{"iopub.status.busy":"2022-07-25T23:57:02.705990Z","iopub.execute_input":"2022-07-25T23:57:02.706723Z","iopub.status.idle":"2022-07-25T23:57:02.726999Z","shell.execute_reply.started":"2022-07-25T23:57:02.706688Z","shell.execute_reply":"2022-07-25T23:57:02.725496Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# px.scatter(pred_data,x=x,y=[\"predicted_SalePrice\"])","metadata":{"execution":{"iopub.status.busy":"2022-07-25T23:57:02.728950Z","iopub.execute_input":"2022-07-25T23:57:02.729661Z","iopub.status.idle":"2022-07-25T23:57:02.735081Z","shell.execute_reply.started":"2022-07-25T23:57:02.729615Z","shell.execute_reply":"2022-07-25T23:57:02.733854Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Say our Ground Living Area was 348 square feet, then:\n$$SalePrice = 18569 + (107 * 348) = \\$18569+37236 = \\$55805$$","metadata":{"execution":{"iopub.status.busy":"2022-07-01T01:12:45.112079Z","iopub.execute_input":"2022-07-01T01:12:45.112458Z","iopub.status.idle":"2022-07-01T01:12:45.119515Z","shell.execute_reply.started":"2022-07-01T01:12:45.112428Z","shell.execute_reply":"2022-07-01T01:12:45.11813Z"}}},{"cell_type":"code","source":"intercept = simp_lin_model.params[0]\nslope = simp_lin_model.params[1]\n\narea = 348\nsale_price = intercept + slope * area\nsale_price","metadata":{"execution":{"iopub.status.busy":"2022-07-25T23:57:02.737083Z","iopub.execute_input":"2022-07-25T23:57:02.737865Z","iopub.status.idle":"2022-07-25T23:57:02.762554Z","shell.execute_reply.started":"2022-07-25T23:57:02.737815Z","shell.execute_reply":"2022-07-25T23:57:02.761256Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"simp_lin_model.resid","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"execution":{"iopub.status.busy":"2022-07-25T23:57:02.764513Z","iopub.execute_input":"2022-07-25T23:57:02.765286Z","iopub.status.idle":"2022-07-25T23:57:02.783573Z","shell.execute_reply.started":"2022-07-25T23:57:02.765236Z","shell.execute_reply":"2022-07-25T23:57:02.782340Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"simp_lin_model.mse_resid","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"execution":{"iopub.status.busy":"2022-07-25T23:57:02.785366Z","iopub.execute_input":"2022-07-25T23:57:02.790586Z","iopub.status.idle":"2022-07-25T23:57:02.802749Z","shell.execute_reply.started":"2022-07-25T23:57:02.790529Z","shell.execute_reply":"2022-07-25T23:57:02.801154Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"np.sqrt(simp_lin_model.mse_resid)","metadata":{"execution":{"iopub.status.busy":"2022-07-25T23:57:02.805000Z","iopub.execute_input":"2022-07-25T23:57:02.805793Z","iopub.status.idle":"2022-07-25T23:57:02.815055Z","shell.execute_reply.started":"2022-07-25T23:57:02.805745Z","shell.execute_reply":"2022-07-25T23:57:02.813782Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"simp_lin_model.get_influence().summary_frame()","metadata":{"execution":{"iopub.status.busy":"2022-07-25T23:57:02.817029Z","iopub.execute_input":"2022-07-25T23:57:02.817980Z","iopub.status.idle":"2022-07-25T23:57:03.680228Z","shell.execute_reply.started":"2022-07-25T23:57:02.817931Z","shell.execute_reply":"2022-07-25T23:57:03.679201Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"simp_lin_model.summary()","metadata":{"execution":{"iopub.status.busy":"2022-07-25T23:57:03.681678Z","iopub.execute_input":"2022-07-25T23:57:03.682003Z","iopub.status.idle":"2022-07-25T23:57:03.701160Z","shell.execute_reply.started":"2022-07-25T23:57:03.681974Z","shell.execute_reply":"2022-07-25T23:57:03.700403Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"lin_reg = LinearRegression()\nlin_reg_scores = cross_val_score(lin_reg,X_train_prepped,y_train,\n                                 scoring=\"neg_mean_squared_error\",\n                                 cv=sk_split.split(X_train_prepped,y_train))\n\nlin_rmse_scores = np.sqrt(-lin_reg_scores)\nprint(f'Average RMSE Score : {np.mean(lin_rmse_scores)}')","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"execution":{"iopub.status.busy":"2022-07-25T23:57:03.702254Z","iopub.execute_input":"2022-07-25T23:57:03.702968Z","iopub.status.idle":"2022-07-25T23:57:03.723047Z","shell.execute_reply.started":"2022-07-25T23:57:03.702936Z","shell.execute_reply":"2022-07-25T23:57:03.722002Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"best_estimators.add((\"Linear Regression\",lin_reg))","metadata":{"execution":{"iopub.status.busy":"2022-07-25T23:57:03.724411Z","iopub.execute_input":"2022-07-25T23:57:03.724799Z","iopub.status.idle":"2022-07-25T23:57:03.729166Z","shell.execute_reply.started":"2022-07-25T23:57:03.724768Z","shell.execute_reply":"2022-07-25T23:57:03.728375Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Ridge Regression","metadata":{}},{"cell_type":"markdown","source":"Ridge Regression is a regularized version of Linear Regression. The Ridge regression model fits all the predictors $p$ and constrains/regularizies the coefficent estimates towards zero. Shrinking the coefficent estimates towards zero can reduce their variance.\n\n$$RSS+\\alpha \\sum\\limits_{i=1}^{n}b_i^2$$\n\n$\\alpha \\geq 0$ is known as a tuning parameter, it controls how much you want to regularize the model. \n\n$\\alpha \\sum\\limits_{i=1}^{n}b_i^2$ is called the shrinkage penalty ($l2$ regularization).\nWhen $\\alpha = 0$, their is no effect from the penalty term, and ridge regression will produce a least squares estimate, however as the penalty gets larger, the impact of the penalty grows and the coefficent estimates will approach zero.\n\nIf $X$ has been centered to have mean zero before ridge regression is performed, then the estimated intercept will take the form: $b_0 = \\bar{y}=\\sum\\limits_{i=1}^{n}\\frac{y_i}{n}$\n\n### Improvements over least squares\nFirst, we must look at the **bias - variance trade off**:\n\n**Bias** refers to the error that is introduced by approximating a real world practical problem (usually a complex problem), by using a simpler model. One such case is assuming the data is linear when it is a polynomial of some $nth$ degree, as such it will not be able to produce an accurate estimate using the simple model no matter how much observations it is fed.  This will lead to the model underfitting the data, a higher bias model means it is more likely to underfit the data. \n\n**Variance** refers to the amount by which the model prediction would change when it is examined on an unseen data set. Differing training sets would result in differing model predictions, and ideally the model predictions would not vary too much among the sets. A model with high variance is susceptible to small changes in the training data, which can lead to vastly different model predictions. Essentially, the model with high variance has a complex fit to the training model and will not be able to accurately fit new unseen data, this is referred to as overfitting the data.\n\nGenerally speaking, simpler models have higher bias and lower variance whereas more complex models have greater variance but lower bias, this is the tradeoff. Ideally, a good model would be one that has low variance and low squared bias.\n\nWith respect to ridge regression, as $\\alpha$ increases, the ridge regression fit decreases, i.e lowering the variance, but increasing the bias. \n\nIt is important to scale the data before performing Ridge regression as it is sensitive to the scale of the input data","metadata":{}},{"cell_type":"code","source":"ridge_reg = Ridge()\n\nparams = {\"alpha\":np.linspace(0.0001,1.0,10),\n          \"fit_intercept\":[True,False],\n          \"solver\":['auto', 'svd', 'cholesky', 'lsqr', 'sparse_cg', 'sag', 'saga', 'lbfgs']}\nridge_gs = GridSearchCV(ridge_reg, param_grid=params,refit=True, n_jobs=-1,scoring = \"neg_mean_squared_error\",cv=sk_split).fit(X_train_prepped,y_train)","metadata":{"execution":{"iopub.status.busy":"2022-07-25T23:57:03.730393Z","iopub.execute_input":"2022-07-25T23:57:03.731160Z","iopub.status.idle":"2022-07-25T23:57:06.105397Z","shell.execute_reply.started":"2022-07-25T23:57:03.731130Z","shell.execute_reply":"2022-07-25T23:57:06.104008Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"best_estimators.add((\"Ridge Regression\",Ridge(**ridge_gs.best_params_)))","metadata":{"execution":{"iopub.status.busy":"2022-07-25T23:57:06.107229Z","iopub.execute_input":"2022-07-25T23:57:06.107690Z","iopub.status.idle":"2022-07-25T23:57:06.114026Z","shell.execute_reply.started":"2022-07-25T23:57:06.107642Z","shell.execute_reply":"2022-07-25T23:57:06.112964Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Lasso Regression","metadata":{}},{"cell_type":"markdown","source":"Least Absolute Shrinkage and Selection Operator Regression (Lasso) is another form of regularized Linear Regression (uses $l1$ regularization).\n\n$$RSS +\\alpha \\sum\\limits_{i=1}^{n}|b_i|$$\n\nOne of the main drawbacks with Ridge Regression, is that all the predictors will be included in the final model, the penalty may shrinks all the coeffiecnts near to zero but never actually be set to zero. With Lasso Regression, Lasso performs variable selection, as the $l1$ penalty has the effect of setting some of the coefficent estimates to zero, if $\\alpha$ is large enough. This means only a subset of variables is yielded in the final model.\n\nLasso can also be used as a method of feature selection","metadata":{}},{"cell_type":"code","source":"lasso_reg = Lasso(random_state=RANDOM_STATE)\n\nparams = {\"alpha\":np.linspace(0.0001,1.0,100),\n          \"fit_intercept\":[True,False]}\nlasso_gs = GridSearchCV(lasso_reg,\n                        param_grid=params,\n                        refit=True,\n                        n_jobs=-1,\n                        scoring = \"neg_mean_squared_error\",\n                        cv=sk_split).fit(X_train_prepped,y_train)","metadata":{"execution":{"iopub.status.busy":"2022-07-25T23:57:06.115464Z","iopub.execute_input":"2022-07-25T23:57:06.116392Z","iopub.status.idle":"2022-07-25T23:57:06.946120Z","shell.execute_reply.started":"2022-07-25T23:57:06.116357Z","shell.execute_reply":"2022-07-25T23:57:06.945015Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"best_estimators.add((\"Lasso Regression\",Lasso(**lasso_gs.best_params_)))","metadata":{"execution":{"iopub.status.busy":"2022-07-25T23:57:06.947743Z","iopub.execute_input":"2022-07-25T23:57:06.948410Z","iopub.status.idle":"2022-07-25T23:57:06.953356Z","shell.execute_reply.started":"2022-07-25T23:57:06.948374Z","shell.execute_reply":"2022-07-25T23:57:06.952357Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Elastic Net","metadata":{}},{"cell_type":"markdown","source":"Elastic Net is sort of a middle ground between Ridge Regression and Lasso Regression. It uses a mix of $l1$ regularization and $l2$ regularization\n\n$$RSS +r\\alpha \\sum\\limits_{i=1}^{n}|b_i| + \\frac{1-r}{2}\\alpha\\sum\\limits_{i=1}^{n}b_i^2$$","metadata":{}},{"cell_type":"code","source":"elastic_reg = ElasticNet(random_state=RANDOM_STATE)\n\nparams = {\"alpha\":np.linspace(0.0001,1.0,10),\"fit_intercept\":[True,False],\"l1_ratio\":np.linspace(0.01,1,50)}\nelastic_gs = GridSearchCV(elastic_reg,\n                        param_grid=params,\n                        refit=True,\n                        n_jobs=-1,\n                          scoring = \"neg_mean_squared_error\",\n                          cv=sk_split).fit(X_train_prepped,y_train)","metadata":{"execution":{"iopub.status.busy":"2022-07-25T23:57:06.955059Z","iopub.execute_input":"2022-07-25T23:57:06.955498Z","iopub.status.idle":"2022-07-25T23:57:09.775988Z","shell.execute_reply.started":"2022-07-25T23:57:06.955455Z","shell.execute_reply":"2022-07-25T23:57:09.775147Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"best_estimators.add((\"Elastic Net Regression\",ElasticNet(**elastic_gs.best_params_)))","metadata":{"execution":{"iopub.status.busy":"2022-07-25T23:57:09.777517Z","iopub.execute_input":"2022-07-25T23:57:09.778104Z","iopub.status.idle":"2022-07-25T23:57:09.782045Z","shell.execute_reply.started":"2022-07-25T23:57:09.778071Z","shell.execute_reply":"2022-07-25T23:57:09.781369Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Gradient Descent\nThis is a generic optimization algorithm that is used to minimize a cost function (eg. RSS, MSE etc) in an iterative manner by moving towards the steepest descent, which is defined by the negative of the gradient, until the algorithm converges to a minimum (gradient is zero). The weights are randomly initialized and then improved upon gradually with each iteration decreasing the cost function.\n\n![Gradient Descent](https://miro.medium.com/max/1400/1*f0CuPDSWFUr9XGESWQ4JUA.png)\n\nOne of the main hyperparameters of discussion in Gradient Descent is the **learning rate** hyperparamerter, i.e the size of the steps. The learning step size is proportional to the slope of the cost function, so it the steps get smaller as the minimum is approached.\n\nIn the case of the learning rate being too high, this may cause the algorithm to diverge and possibly unable to find an optimal solution, it can inadvertently increase the training error.  If the learning rate is too small, training may take too long to converge.\n\nA pitfall with gradient descent is that some cost functions have irregular shapes, With reference to the diagram below, if the random initialzation starts out on the left, then it converges to a local minimum, which is obviously not as good as the global minimum.\n\n![Local Min Max](https://www.researchgate.net/profile/Leandro-De-Castro/publication/2428674/figure/fig2/AS:669408926130190@1536610933542/Scalar-example-of-a-function-with-one-local-and-the-global-minimum.png)\n\nLearning rates cannot be calculated analytically and must be done via trial and error. A good starting point is 0.1 or 0.01. The values typically range from 1.0 to 0.000001.\n\nFeatures should be scaled before using gradient descent.\n\n## Batch Gradient Descent \nThe gradient of the cost function is computed with respect to each model parameter, $b_i$, this is also known as taking the partial derivative of the cost function. Once the gradient vector is calculated, we then take this gradient vector and multiply it by the learning rate (this is the amount of step the weight will take downwards). Finally we subtract this step from the current weights, to get the the weights for the next step.\n\nWhy is it called Batch Gradient Descent? Because the gradient vector is calculated over the full training set at each step and as such it is slower on larger training sets. \n\nWe may need to run Batch Gradient Descent more than once, with a different starting point each time and then comparing the performance on a validation set.\n## Stochastic Gradient Descent (SGD)\nOn the other side of the spectrum, there is *Stochastic Gradient Descent*, instead of computing the gradients over the entire training set, it picks a single *random* instance from the training set and computes the gradient for that single instance. This update is repeated by cycling through the data. Due to it's random nature, it is not as regular as Batch Gradient Descent and as such may reach close the global minimum but never settle there. Also, the randomness can help to escape from the local minimum whenever the cost function is highly irregular.\n\nA solution to SGD not settling at the global minimum is to gradually reduce the learning rate, the function that determines the the learning rate is called the learning schedule. Reducing the learning rate too fast may leave it stuck at some local minimum, of if reduced too slowly, it may bounce around and end up at some suboptimal solution if training is halted early.\n\nWith SGD, we need to ensure the training instances are independent and identically distributed, as to ensure a global minimum is reached. A common apprach to deal with this issue is to shuffle the training set (features and labels together) at the beginning of each epoch.\n## Mini-batch Gradient Descent\nThe gradients are computed on small randomly selected sets of instances from the training data, which garners the name *mini-batch*. It is more stable than SGD and gets closer to the global minimum but may have difficulty escaping a local minimum with highly irregular cost functions.\n\n[img src](https://miro.medium.com/max/1400/1*f0CuPDSWFUr9XGESWQ4JUA.png)","metadata":{}},{"cell_type":"code","source":"# sgd_reg = SGDRegressor(tol=1e-3,\n#                        shuffle=True,\n#                        random_state=RANDOM_STATE)\n# sgd_reg.fit(X_train_prepped,y_train)\n# print(sgd_reg.score(X_train_prepped,y_train))\n# y_pred = sgd_reg.predict(X_train_prepped)\n# np.sqrt(MSE(y_train_reshaped,y_pred))\n\n# params = [\n# #     {\"learning_rate\":[\"constant\",\"adaptive\"],\n# #          \"penalty\":[\"l1\",\"l2\",\"elasticnet\",None],\n# #          \"eta0\":np.logspace(-6, 0, 20),\n# #           \"alpha\":np.linspace(0.0001,1.0,10),\n# #            \"fit_intercept\":[True,False],\n# #           \"l1_ratio\":np.linspace(0.1,1,50)},\n         \n# #           {\"learning_rate\":[\"optimal\"],\n# #          \"penalty\":[\"l1\",\"l2\",\"elasticnet\",None],\n# #          \"alpha\":np.linspace(0.0001,1.0,10),\n# #          \"fit_intercept\":[True,False],\n# #          \"l1_ratio\":np.linspace(0.1,1,50)},\n          \n#          {\"learning_rate\":[\"invscaling\"],\n#          \"penalty\":[\"l1\",\"l2\",\"elasticnet\",None],\n#          \"eta0\":np.logspace(-6, 6, 50),\n#           \"l1_ratio\":np.linspace(0.1,1,50),\n#          \"power_t\":np.linspace(0,1,50),\n#          \"alpha\":np.linspace(0.0001,1.0,50),\n#          \"fit_intercept\":[True,False]}]\n\n\n# sgd_gs = RandomizedSearchCV(sgd_reg,\n#                         param_distributions=params,\n#                         n_iter=1000,\n#                         refit=True,\n#                         n_jobs=-1,scoring = \"neg_mean_squared_error\",cv=sk_split).fit(X_train_prepped,y_train)\n# sgd_best = sgd_gs.best_estimator_\n# print(sgd_gs.best_params_)\n# print(sgd_best.score(X_train_prepped,y_train))\n# print(sgd_best.coef_)\n# sgd_cv_results = pd.DataFrame(sgd_gs.cv_results_)","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-07-25T23:57:09.783375Z","iopub.execute_input":"2022-07-25T23:57:09.783899Z","iopub.status.idle":"2022-07-25T23:57:09.792890Z","shell.execute_reply.started":"2022-07-25T23:57:09.783869Z","shell.execute_reply":"2022-07-25T23:57:09.792222Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Polynomial Regression\n\nThis can be thought of as an extension to the linear model. The linear model is exttended by adding more predictors by raising the original predictors to a power. Polynomial regression, when specified to some degree, $d$  allows us to generate a non - linear curve.\n\nA Polynomial function generally looks like this:\n$$y_i = b_0+b_1X_i+b_2X_i^2+b_3X_i^3+b_4X_i^4+...+b_dX_i^d+\\epsilon_i$$\n\nPolynomials of degree $d$, generally don't exceed $d=3,d=4$ because overfitting may starting taking place.\n\nPolynomial features of some degree $d$ transforms $n$ features into $\\frac{(n+d)!}{d!n!}$ features.\n\nSo, if we were to take our $8$ features and transform it using a polynomial regression of degree $d=3$, we would get\n\n$$\\frac{(8+3)!}{8!3!} = \\frac{39,916,800}{241,920} = 165 features $$\n\nHowever, if we were take a polynomial of degree $d=10$, \n\n$$\\frac{(8+10)!}{8!10!} = \\frac{6.402373705728E15}{146,313,216,000} = 43,758 features $$\n\nJust for fun, if we take, degree $d=100$, we would get, \n\n$$\\frac{(8+100)!}{8!100!} = 352,025,629,371 features $$\n\n**A cool, 350 billion features.**","metadata":{}},{"cell_type":"code","source":"# To play nice with MSE,MAE\ny_train_reshaped = y_train.to_numpy().reshape((-1,1))","metadata":{"execution":{"iopub.status.busy":"2022-07-25T23:57:09.795970Z","iopub.execute_input":"2022-07-25T23:57:09.796309Z","iopub.status.idle":"2022-07-25T23:57:09.808802Z","shell.execute_reply.started":"2022-07-25T23:57:09.796279Z","shell.execute_reply":"2022-07-25T23:57:09.807750Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### [Along Came Polly](https://youtu.be/_tsUZbkynY8)","metadata":{}},{"cell_type":"code","source":"plain_estimators = [(\"Linear Regression\",LinearRegression()),\n                    (\"Ridge Regression\",Ridge()),\n                   (\"Lasso Regression\",Lasso()),\n                   (\"Elastic Net Regression\",ElasticNet())]","metadata":{"execution":{"iopub.status.busy":"2022-07-25T23:57:09.810253Z","iopub.execute_input":"2022-07-25T23:57:09.810886Z","iopub.status.idle":"2022-07-25T23:57:09.818672Z","shell.execute_reply.started":"2022-07-25T23:57:09.810845Z","shell.execute_reply":"2022-07-25T23:57:09.817684Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"name_list = []\ndegree_list = []\nMSE_list =[]\nMAE_list =[]\nR2_list =[]\n\npoly_df = pd.DataFrame(columns=[\"Name\",\"Degree\",\"MSE\",\"MAE\",\"R2\"])","metadata":{"execution":{"iopub.status.busy":"2022-07-25T23:57:09.820010Z","iopub.execute_input":"2022-07-25T23:57:09.820582Z","iopub.status.idle":"2022-07-25T23:57:09.829275Z","shell.execute_reply.started":"2022-07-25T23:57:09.820542Z","shell.execute_reply":"2022-07-25T23:57:09.828537Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Check up to polynomial degree 5, however, if you increase the components just even by a little, it can have a exploding effect. (It's commented out when the number of components is more than say 15.","metadata":{}},{"cell_type":"code","source":"# for degree in [2,3,4,5,6]:\n# for degree in [2,3,4,5]:\nfor degree in [2,3]:\n    for tup in plain_estimators:\n        name = tup[0]\n        estimator = tup[1]\n        along_came_polly = PolynomialFeatures(degree,include_bias = False)\n        X_poly = along_came_polly.fit_transform(X_train_prepped)\n        estimator.fit(X_poly,y_train)\n        \n        name_list.append(name)\n        degree_list.append(degree)\n        MSE_list.append(np.sqrt(MSE(y_train_reshaped,estimator.predict(X_poly))))\n        MAE_list.append(MAE(y_train_reshaped,estimator.predict(X_poly)))\n        R2_list.append(estimator.score(X_poly,y_train))","metadata":{"execution":{"iopub.status.busy":"2022-07-25T23:57:09.831006Z","iopub.execute_input":"2022-07-25T23:57:09.831417Z","iopub.status.idle":"2022-07-25T23:57:09.844082Z","shell.execute_reply.started":"2022-07-25T23:57:09.831379Z","shell.execute_reply":"2022-07-25T23:57:09.843232Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"poly_df[\"Name\"] = name_list\npoly_df[\"Degree\"] = degree_list\npoly_df[\"MSE\"] = MSE_list\npoly_df[\"MAE\"] = MAE_list\npoly_df[\"R2\"] = R2_list","metadata":{"execution":{"iopub.status.busy":"2022-07-25T23:57:09.846328Z","iopub.execute_input":"2022-07-25T23:57:09.847278Z","iopub.status.idle":"2022-07-25T23:57:09.854966Z","shell.execute_reply.started":"2022-07-25T23:57:09.847245Z","shell.execute_reply":"2022-07-25T23:57:09.854284Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"poly_df","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2022-07-25T23:57:09.856217Z","iopub.execute_input":"2022-07-25T23:57:09.856715Z","iopub.status.idle":"2022-07-25T23:57:09.865677Z","shell.execute_reply.started":"2022-07-25T23:57:09.856684Z","shell.execute_reply":"2022-07-25T23:57:09.864916Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Polynomial MSE Comparison","metadata":{}},{"cell_type":"code","source":"fig = px.histogram(poly_df,x=\"Name\",y=[\"MSE\"],color=\"Degree\",barmode=\"group\",histfunc=\"sum\")\nfig.update_layout(title=\"Comparision of MSE using varying polynomial degrees (Lower is better)\")\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-25T23:57:09.866906Z","iopub.execute_input":"2022-07-25T23:57:09.867422Z","iopub.status.idle":"2022-07-25T23:57:09.877268Z","shell.execute_reply.started":"2022-07-25T23:57:09.867391Z","shell.execute_reply":"2022-07-25T23:57:09.876518Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Polynomial MAE Comparison","metadata":{}},{"cell_type":"code","source":"fig = px.histogram(poly_df,x=\"Name\",y=[\"MAE\"],color=\"Degree\",barmode=\"group\",histfunc=\"sum\")\nfig.update_layout(title=\"Comparision of MAE using varying polynomial degrees (Lower is better)\")\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-25T23:57:09.878434Z","iopub.execute_input":"2022-07-25T23:57:09.878736Z","iopub.status.idle":"2022-07-25T23:57:09.888057Z","shell.execute_reply.started":"2022-07-25T23:57:09.878708Z","shell.execute_reply":"2022-07-25T23:57:09.887005Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Best possible score is 1.0 and it can be negative (because the model can be arbitrarily worse). In the general case when the true y is non-constant, a constant model that always predicts the average y disregarding the input features would get a score of 0.0.","metadata":{}},{"cell_type":"markdown","source":"## Polynomial $R^2$ Comparison","metadata":{}},{"cell_type":"code","source":"fig = px.histogram(poly_df,x=\"Name\",y=[\"R2\"],color=\"Degree\",barmode=\"group\",histfunc=\"sum\")\nfig.update_layout(title=\"Comparision of R2 using varying polynomial degrees\")\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-25T23:57:09.889253Z","iopub.execute_input":"2022-07-25T23:57:09.890116Z","iopub.status.idle":"2022-07-25T23:57:09.900705Z","shell.execute_reply.started":"2022-07-25T23:57:09.890083Z","shell.execute_reply":"2022-07-25T23:57:09.899745Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# KNN Regressor","metadata":{}},{"cell_type":"code","source":"knn = KNeighborsRegressor()\nparams=[{\"n_neighbors\":np.arange(1,100,2),\n        \"algorithm\":[\"auto\", \"ball_tree\", \"kd_tree\", \"brute\"],\n       \"weights\":[\"uniform\",\"distance\"]}]\nknn_cv = GridSearchCV(knn,\n                      params,\n                      refit=True,\n                      scoring = \"neg_mean_squared_error\",\n                      cv=sk_split).fit(X_train_prepped,y_train )","metadata":{"execution":{"iopub.status.busy":"2022-07-25T23:57:09.902237Z","iopub.execute_input":"2022-07-25T23:57:09.902820Z","iopub.status.idle":"2022-07-25T23:57:18.765509Z","shell.execute_reply.started":"2022-07-25T23:57:09.902789Z","shell.execute_reply":"2022-07-25T23:57:18.764394Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"best_estimators.add((\"KNN Regressor\",KNeighborsRegressor(**knn_cv.best_params_)))","metadata":{"execution":{"iopub.status.busy":"2022-07-25T23:57:18.766856Z","iopub.execute_input":"2022-07-25T23:57:18.767207Z","iopub.status.idle":"2022-07-25T23:57:18.771753Z","shell.execute_reply.started":"2022-07-25T23:57:18.767156Z","shell.execute_reply":"2022-07-25T23:57:18.770692Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Random Forest\n<!-- ![Decision Tree & Random Forest Side by Side](https://upload.wikimedia.org/wikipedia/commons/d/d8/Decision_Tree_vs._Random_Forest.png?20201214034417)\n[img src](https://commons.wikimedia.org/wiki/File:Decision_Tree_vs._Random_Forest.png) -->\n\nIn order to talk about Random Forests, we must first talk about Decision trees.\n\n## Decision Trees \n\nLooking at *Regression Trees* in particular. \n\nTree-based methods involve stratifying / segmenting the predictor space into many simpler regions. In making a prediction, we use some sort of metric such as the mean or median of the training data in it's respective region. \n\nConstructing a decision tree:\n\n* Take the features (predictors), that is, take $X_1, X_2,...,X_p$ and partition into $J$ distinct and non overalpping regions, $R_1,R_2,...,R_j$. \n* Every observation that  falls into some region, $R_j$, the same prediction is made by taking some metric, such as the mean or median of the values from the training data that falls into the region $R_j$.\n\nTheoretically, the regions can have any shape (boundary) but are usually bounded into boxes/rectangles for ease of interpretability. As with the previous modelling methods, we want to minimize $RSS$, which is given by\n\n$$\\sum\\limits_{j=1}^{J}\\sum\\limits_{i \\in R_j}(y_i - \\hat{y}_{R_j})^2$$ where $\\hat{y}_{R_j}$ is the mean response for the training observations within the $jth$ box.\n\nIn determining the partitions of the feature space into $J$ boxes, an approach called *recursive binary splitting*. In general when performing recursive binary splitting, the first feature $X_j$ and a split point $s$ is selected, this split point is used to split the feature/predictor space into two regions, $\\{X|X_j <s\\}$ and $\\{X|X_j \\geq s\\}$. In choosing a specified predictor/feature and split point, all features and split points are considered and then chosen based on which the final tree would minimize the $RSS$. This process is then repeated by again looking for the best predictor and split point in order to split the data further until some specified stopping criterion is reached. \n\nA large tree may be susceptible to overfitting the data whereas a small tree may not be able to capture the entire complex structure of the data. \n\nThe main idea of **Pruning** a regression tree is prevent overfitting of the data.\n\nA strategy used to prune a tree, is called *Cost complexity pruning* also known as *weakest link pruning*. The main idea behind this strategy is to grow a very large tree and start pruning backwards to get a subtree. The subtree is selected based on the lowest test error rate (usually done by cross validation), however it may not be possible to check all trees and therefore only a subset of the subtrees will be considered. \n\nIf we consider a sequence of trees, with a nonnegative hyperparameter $\\alpha$, for each value of $\\alpha$ there is a corresponding subtree $T \\subset T_0$ (where $T_0$ is the entire tree) such that \n\n$$\\sum\\limits_{m=1}^{|T|}\\sum\\limits_{i:x_i \\in R_m}(y_i - \\hat{y}_{R_m})^2 +\\alpha|T|$$\n\nis as small as possiible.\n\n* $|T|$ is the number of terminal nodes of the tree $T$\n* $R_m$ is the rectangle (subset of the predictor space) with the associated $mth$ terminal node\n* $\\hat{y}_{R_m}$ is the predicted response with the associated $R_m$\n\nNote: If $\\alpha = 0$, then $T=T_0$\n\n## Bagging & Pasting\n\n*Bagging* short for Bootstrap Aggregration, is a way in which to reduce the variance of an estimator. If we have a set of observations (samples) $X_1,...,X_n$ each with their own variance $\\sigma^2$, then the average of these observations is $\\sigma/n$, this means the averaging of the observations reduces variance. If sampling without replacement is done, this is called *pasting*. These methods of sampling are of particular use in Decision Trees. \n\nWe can use multiple decision trees (hundreds or even thousands) when using the bagging method, this allows the training data to be sampled several times across multiple predictors. These trees are grown deep and not pruned.  Once trained, a final prediction is made by aggregrating all the predictions from the predictors, usually this is done by the most frequent prediction. \n\n### Out of the bag (OOB) Evaluation\nIt is estimated that a bagged tree, uses around two thirds of the training observations. The remaining one third of the training observations that aren't used are called *out of the bag* instances and the predictor can be evaluated using these instances and thus get an average out of all the predictors. \n\n## Random Forests\nRandom forests are considered an improvement bagged trees, in that it decorrelates the trees. When splitting the tree, only a *random sample* of predictors from the full set of predictors are chosen as the split point options. In essence, instead of searching for the best feature/predictor to split a node, it searches a subset of features/predictors to split on, typically this subset is of size $\\sqrt{p}$.\n\nIn bagging, since is uses all the predictors, it usually will pick the strongest predictors to split on and as such, the trees look similar and thus be highly correlated.This will not lead to a substantial reduction in variance.  With random forests, since it does not consider all the predictors for a split point, other predictors have more of a chance to become that split point. This reduces the correlation and thus the average of the tress will lead to lower variance.\n\n### Extra Trees\nAn extra layer of randomness can be added to RandomForests by using random thresholds for each feature.","metadata":{}},{"cell_type":"markdown","source":"<!-- ![](https://images.unsplash.com/photo-1593069567131-53a0614dde1d?ixlib=rb-1.2.1&ixid=MnwxMjA3fDB8MHxwaG90by1wYWdlfHx8fGVufDB8fHx8&auto=format&fit=crop&w=1632&q=80) -->","metadata":{}},{"cell_type":"markdown","source":"### Simple Decision Tree\nIf we keep increasing the depth of the single tree, we see that the $RMSE$ score approaches zero and the $R^2$ approaches 1, while this might seem like a great thing to have, this is usually a sign of overfitting the data. ","metadata":{}},{"cell_type":"code","source":"tree_reg = DecisionTreeRegressor(max_depth = 10)\ntree_reg.fit(X_train_prepped,y_train)\nprint(tree_reg.score(X_train_prepped,y_train))\ny_pred = tree_reg.predict(X_train_prepped)\nnp.sqrt(MSE(y_train_reshaped,y_pred))","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2022-07-25T23:57:18.773125Z","iopub.execute_input":"2022-07-25T23:57:18.774109Z","iopub.status.idle":"2022-07-25T23:57:18.802561Z","shell.execute_reply.started":"2022-07-25T23:57:18.774067Z","shell.execute_reply":"2022-07-25T23:57:18.801428Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Boostrap","metadata":{}},{"cell_type":"code","source":"# bag_reg_boot = BaggingRegressor(DecisionTreeRegressor(),\n#                                 bootstrap=True,\n#                                random_state=RANDOM_STATE)\n\n# params = [{\"n_estimators\":np.arange(1000,2000+1,200),\n#            \"max_samples\":np.arange(10,200,20)}]\n\n# bag_reg_boot_cv = RandomizedSearchCV(bag_reg_boot,\n#                       params,\n#                       refit=True,\n#                         n_iter = 10,\n#                       scoring = \"neg_mean_squared_error\",\n#                       cv=sk_split).fit(X_train_prepped,y_train )","metadata":{"execution":{"iopub.status.busy":"2022-07-25T23:57:18.805108Z","iopub.execute_input":"2022-07-25T23:57:18.805927Z","iopub.status.idle":"2022-07-25T23:58:27.778490Z","shell.execute_reply.started":"2022-07-25T23:57:18.805884Z","shell.execute_reply":"2022-07-25T23:58:27.777507Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# best_estimators.add((\"Bagging Regressor\",BaggingRegressor(DecisionTreeRegressor(),**bag_reg_boot_cv.best_params_)))","metadata":{"execution":{"iopub.status.busy":"2022-07-25T23:58:27.779808Z","iopub.execute_input":"2022-07-25T23:58:27.780134Z","iopub.status.idle":"2022-07-25T23:58:27.785820Z","shell.execute_reply.started":"2022-07-25T23:58:27.780102Z","shell.execute_reply":"2022-07-25T23:58:27.784630Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Pasting","metadata":{}},{"cell_type":"code","source":"# bag_reg_pasting = BaggingRegressor(DecisionTreeRegressor(),\n#                                 bootstrap=False,\n#                                random_state=RANDOM_STATE)\n\n# params = [{\"n_estimators\":np.arange(1000,2000+1,200),\n#            \"max_samples\":np.arange(10,200,20)}]\n\n# bag_reg_pasting_cv = RandomizedSearchCV(bag_reg_pasting,\n#                       params,\n#                       refit=True, n_iter = 10,\n#                       scoring = \"neg_mean_squared_error\",\n#                       cv=sk_split).fit(X_train_prepped,y_train )\n# print(bag_reg_pasting_cv.best_params_)","metadata":{"execution":{"iopub.status.busy":"2022-07-25T23:58:27.787822Z","iopub.execute_input":"2022-07-25T23:58:27.788246Z","iopub.status.idle":"2022-07-25T23:59:36.114253Z","shell.execute_reply.started":"2022-07-25T23:58:27.788204Z","shell.execute_reply":"2022-07-25T23:59:36.113280Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# best_estimators.add((\"Pasting Regressor\",BaggingRegressor(DecisionTreeRegressor(),bootstrap=False,random_state=RANDOM_STATE,\n#                                                           **bag_reg_pasting_cv.best_params_)))","metadata":{"execution":{"iopub.status.busy":"2022-07-25T23:59:36.115687Z","iopub.execute_input":"2022-07-25T23:59:36.116543Z","iopub.status.idle":"2022-07-25T23:59:36.122968Z","shell.execute_reply.started":"2022-07-25T23:59:36.116495Z","shell.execute_reply":"2022-07-25T23:59:36.121914Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Random Patches ","metadata":{}},{"cell_type":"code","source":"bag_reg_patches = BaggingRegressor(DecisionTreeRegressor(),\n                                bootstrap=True,\n                               random_state=RANDOM_STATE)\n\nparams = [{\"n_estimators\":np.arange(1000,2000+1,200),\n           \"max_samples\":np.linspace(0.1,0.9,10),\n         \"max_features\":np.linspace(0.1,1.0,10)}]\n\nbag_reg_patches_cv = RandomizedSearchCV(bag_reg_patches,\n                      params,\n                      refit=True,\n                      scoring = \"neg_mean_squared_error\",\n                      cv=sk_split).fit(X_train_prepped,y_train )\nprint(bag_reg_patches_cv.best_params_)","metadata":{"execution":{"iopub.status.busy":"2022-07-25T23:59:36.124659Z","iopub.execute_input":"2022-07-25T23:59:36.125392Z","iopub.status.idle":"2022-07-26T00:00:49.482660Z","shell.execute_reply.started":"2022-07-25T23:59:36.125356Z","shell.execute_reply":"2022-07-26T00:00:49.481224Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"best_estimators.add((\"Random Patches Regressor\",BaggingRegressor(DecisionTreeRegressor(), bootstrap=True, random_state=RANDOM_STATE,\n                                                          **bag_reg_patches_cv.best_params_)))","metadata":{"execution":{"iopub.status.busy":"2022-07-26T00:00:49.484399Z","iopub.execute_input":"2022-07-26T00:00:49.485309Z","iopub.status.idle":"2022-07-26T00:00:49.491397Z","shell.execute_reply.started":"2022-07-26T00:00:49.485258Z","shell.execute_reply":"2022-07-26T00:00:49.490189Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Random Subspaces","metadata":{}},{"cell_type":"code","source":"bag_reg_sub_space = BaggingRegressor(DecisionTreeRegressor(),\n                                bootstrap=False,\n                                max_samples = 1.0,\n                                bootstrap_features=True,\n                               random_state=RANDOM_STATE)\n\nparams = [{\"n_estimators\":np.arange(1000,2000+1,200),\n         \"max_features\":np.linspace(0.1,0.9,10)}]\n\nbag_reg_sub_space_cv = RandomizedSearchCV(bag_reg_sub_space,\n                      params,\n                      refit=True,\n                      scoring = \"neg_mean_squared_error\",\n                      cv=sk_split).fit(X_train_prepped,y_train )\nprint(bag_reg_sub_space_cv.best_params_)","metadata":{"execution":{"iopub.status.busy":"2022-07-26T00:00:49.493106Z","iopub.execute_input":"2022-07-26T00:00:49.494339Z","iopub.status.idle":"2022-07-26T00:02:28.718088Z","shell.execute_reply.started":"2022-07-26T00:00:49.494294Z","shell.execute_reply":"2022-07-26T00:02:28.716834Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"best_estimators.add((\"Random Subspaces Regressor\",BaggingRegressor(DecisionTreeRegressor(), bootstrap=False, max_samples = 1.0,bootstrap_features=True,\n                               random_state=RANDOM_STATE, **bag_reg_sub_space_cv.best_params_)))","metadata":{"execution":{"iopub.status.busy":"2022-07-26T00:02:28.719605Z","iopub.execute_input":"2022-07-26T00:02:28.720570Z","iopub.status.idle":"2022-07-26T00:02:28.725109Z","shell.execute_reply.started":"2022-07-26T00:02:28.720533Z","shell.execute_reply":"2022-07-26T00:02:28.724340Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### RandomForests","metadata":{}},{"cell_type":"code","source":"rnd_forest_reg = RandomForestRegressor()\nparams = [{\"n_estimators\":np.arange(1000,2000+1,200),\n           \"criterion\":[\"squared_error\", \"absolute_error\", \"poisson\"],\n          \"max_depth\":np.arange(1,10,1),\n          \"max_features\":[\"sqrt\", \"log2\", None]}]\n\nrnd_forest_reg_cv = RandomizedSearchCV(rnd_forest_reg,params,\n                                   scoring=\"neg_mean_squared_error\",refit=True,\n                                   cv=sk_split).fit(X_train_prepped,y_train)\nprint(rnd_forest_reg_cv.best_params_)","metadata":{"execution":{"iopub.status.busy":"2022-07-26T00:02:28.726370Z","iopub.execute_input":"2022-07-26T00:02:28.727276Z","iopub.status.idle":"2022-07-26T00:04:21.597988Z","shell.execute_reply.started":"2022-07-26T00:02:28.727243Z","shell.execute_reply":"2022-07-26T00:04:21.596908Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"best_estimators.add((\"Random Forest Regressor\",RandomForestRegressor(**rnd_forest_reg_cv.best_params_)))","metadata":{"execution":{"iopub.status.busy":"2022-07-26T00:04:21.599422Z","iopub.execute_input":"2022-07-26T00:04:21.600010Z","iopub.status.idle":"2022-07-26T00:04:21.604187Z","shell.execute_reply.started":"2022-07-26T00:04:21.599978Z","shell.execute_reply":"2022-07-26T00:04:21.603325Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Extra Trees ","metadata":{}},{"cell_type":"code","source":"amazon_reg = ExtraTreesRegressor()\nparams = [{\"n_estimators\":np.arange(100,2000+1,100),\n           \"criterion\":[\"squared_error\", \"absolute_error\", \"poisson\"],\n          \"max_depth\":np.arange(1,10,1),\n          \"max_features\":[\"sqrt\", \"log2\", None]}]\n\namazon_reg_cv = RandomizedSearchCV(amazon_reg,params,\n                                   n_iter=10,\n                                   scoring=\"neg_mean_squared_error\",refit=True,\n                                   cv=sk_split).fit(X_train_prepped,y_train)\nprint(amazon_reg_cv.best_params_)","metadata":{"execution":{"iopub.status.busy":"2022-07-26T00:04:21.605337Z","iopub.execute_input":"2022-07-26T00:04:21.605720Z","iopub.status.idle":"2022-07-26T00:04:59.255480Z","shell.execute_reply.started":"2022-07-26T00:04:21.605690Z","shell.execute_reply":"2022-07-26T00:04:59.254665Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"best_estimators.add((\"Amazon Regressor\",ExtraTreesRegressor(**amazon_reg_cv.best_params_)))","metadata":{"execution":{"iopub.status.busy":"2022-07-26T00:04:59.256520Z","iopub.execute_input":"2022-07-26T00:04:59.257243Z","iopub.status.idle":"2022-07-26T00:04:59.261067Z","shell.execute_reply.started":"2022-07-26T00:04:59.257211Z","shell.execute_reply":"2022-07-26T00:04:59.260327Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Boosting \n*(This topic definitely warrants a deeper dive)*\n\nThe main concept behind boosting is that it combines the outputs of many \"weak\" learners to produce a \"strong\" learner. Estimators are trained sequentially with each trying to correct the previous estimator with information from the previous estimator. ","metadata":{}},{"cell_type":"markdown","source":"## Adaptive Boosting Regression\nWith AdaBoost, a new estimator is \"corrected\" by paying attention to the training instances that the previous estimator underfitted. As such, every new \"weak\" learner (estimator) focus on the examples that were missed by the previous estimator. \n\nIn AdaBoost Regression, a regressor is fit on the training instances and then makes predictions on the training set. The weights are then increased according to the error of the current prediction and are updated accordingly. This is done for all the number of specified estimators. As such, subsequent \"weak\" learners focus more on difficult cases.","metadata":{}},{"cell_type":"code","source":"ada_boost_reg = AdaBoostRegressor(DecisionTreeRegressor(max_depth = 1),\n                                 random_state=RANDOM_STATE)\nparams = [{\"n_estimators\":np.arange(1000,2000+1,200),\n          \"learning_rate\":np.logspace(-5,0,9),\n         \"loss\":[\"linear\", \"square\", \"exponential\"]}]\n\nada_boost_reg_cv = RandomizedSearchCV(ada_boost_reg,\n                                     params,\n                                     refit=True,\n                                     cv=sk_split,\n                                      n_iter = 10,\n                                     scoring=\"neg_mean_squared_error\").fit(X_train_prepped,y_train)","metadata":{"execution":{"iopub.status.busy":"2022-07-26T00:04:59.262245Z","iopub.execute_input":"2022-07-26T00:04:59.262810Z","iopub.status.idle":"2022-07-26T00:06:08.103219Z","shell.execute_reply.started":"2022-07-26T00:04:59.262779Z","shell.execute_reply":"2022-07-26T00:06:08.102107Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"best_estimators.add((\"AdaBoost Regressor\",AdaBoostRegressor(DecisionTreeRegressor(max_depth = 1),\n                                 random_state=RANDOM_STATE,**ada_boost_reg_cv.best_params_)))","metadata":{"execution":{"iopub.status.busy":"2022-07-26T00:06:08.104557Z","iopub.execute_input":"2022-07-26T00:06:08.104878Z","iopub.status.idle":"2022-07-26T00:06:08.110360Z","shell.execute_reply.started":"2022-07-26T00:06:08.104850Z","shell.execute_reply":"2022-07-26T00:06:08.108886Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Gradient Boosting Regression\n\nThe current estimator is fit with the current residuals as the respone. This new estimator is then added and fit in order to update the residuals. Whent the estimators are trees, the trees usually have $d=1$ nodes, i.e **stumps**.\nAs these stumps are fitted to the residuals, the overall predictive accuracy is slowly improved.","metadata":{}},{"cell_type":"code","source":"gd_boost_reg = GradientBoostingRegressor()\nparams = [{\"n_estimators\":np.arange(1000,2000+1,500),\n          \"learning_rate\":np.logspace(-5,1,10),\n         \"loss\":[\"squared_error\", \"absolute_error\"],\n          \"max_depth\":np.arange(1,6,2),\n          \"subsample\":np.linspace(0.1,1.0,3),\n          \"max_features\":[None,\"sqrt\",\"log2\"]},\n          \n         {\"n_estimators\":np.arange(1000,2000+1,500),\n          \"learning_rate\":np.logspace(-5,1,10),\n         \"loss\":[\"huber\", \"quantile\"],\n          \"max_depth\":np.arange(1,6,2),\n         \"alpha\":np.linspace(0.001,0.99,5),\n         \"subsample\":np.linspace(0.1,1.0,3),\n         \"max_features\":[None,\"sqrt\",\"log2\"]}]\n\n\ngd_boost_reg_cv = RandomizedSearchCV(gd_boost_reg,\n                                     params,\n                                     refit=True,\n                                     cv=sk_split,\n                                     n_iter = 15,\n                                     scoring=\"neg_mean_squared_error\").fit(X_train_prepped,y_train)","metadata":{"execution":{"iopub.status.busy":"2022-07-26T00:06:08.113517Z","iopub.execute_input":"2022-07-26T00:06:08.113874Z","iopub.status.idle":"2022-07-26T00:09:29.273212Z","shell.execute_reply.started":"2022-07-26T00:06:08.113843Z","shell.execute_reply":"2022-07-26T00:09:29.272233Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"best_estimators.add((\"Gradient Boosting Regressor\",GradientBoostingRegressor(**gd_boost_reg_cv.best_params_)))","metadata":{"execution":{"iopub.status.busy":"2022-07-26T00:09:29.274385Z","iopub.execute_input":"2022-07-26T00:09:29.274690Z","iopub.status.idle":"2022-07-26T00:09:29.280084Z","shell.execute_reply.started":"2022-07-26T00:09:29.274661Z","shell.execute_reply":"2022-07-26T00:09:29.279040Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## HistGradientBoosting\n\nHistogram-based Gradient Boosting Regression Tree.","metadata":{}},{"cell_type":"code","source":"hist_boost_reg = HistGradientBoostingRegressor(random_state=RANDOM_STATE)\nparams = [{\"learning_rate\":np.logspace(-5,0,9),\n         \"loss\":[\"squared_error\", \"absolute_error\", \"poisson\"],\n          \"l2_regularization\":np.logspace(-3,1,6),\n          \"max_depth\":np.arange(1,10,1)}]\n\nhist_boost_reg_cv = RandomizedSearchCV(hist_boost_reg,\n                                     params,\n                                     refit=True,\n                                     cv=sk_split,\n                                       n_iter = 10,\n                                     scoring=\"neg_mean_squared_error\").fit(X_train_prepped,y_train)","metadata":{"execution":{"iopub.status.busy":"2022-07-26T00:09:29.281549Z","iopub.execute_input":"2022-07-26T00:09:29.282245Z","iopub.status.idle":"2022-07-26T00:09:37.659563Z","shell.execute_reply.started":"2022-07-26T00:09:29.282202Z","shell.execute_reply":"2022-07-26T00:09:37.658639Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"best_estimators.add((\"Histogram Gradient Boosting\",HistGradientBoostingRegressor(random_state=RANDOM_STATE,**hist_boost_reg_cv.best_params_)))","metadata":{"execution":{"iopub.status.busy":"2022-07-26T00:09:37.661089Z","iopub.execute_input":"2022-07-26T00:09:37.661768Z","iopub.status.idle":"2022-07-26T00:09:37.666614Z","shell.execute_reply.started":"2022-07-26T00:09:37.661730Z","shell.execute_reply":"2022-07-26T00:09:37.665582Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Stacking\n\nStacking (Stacked Generalization) has its idea rooted in aggregrating the predictions of all estimators in an ensemble. \n\nStacking allows to use the strength of each individual estimator by using their output as input of a final estimator (called a *blender* or *meta learner*) and thus makes a final prediction.\n\nThe training data is split into two subsets, one for training and another for prediction.  The first subset is used to train the estimators in the first layer and then predictions are made using the hold out set. A new training set is then made out of these predicted values which in turn is used as input features (if $n$ estimators are used, then the new input features will be $n-dimensional$) along with the targets. \n\nThe blender is trained on this new training set and a final prediction is made.\n\n![](https://miro.medium.com/max/1400/0*e-na5r7mF8lVAfPK.png)\n\n[img src](https://cdn-images-1.medium.com/max/809/1*tVq7Ngb3hqY7UJ4DKCxVcg.png)","metadata":{}},{"cell_type":"code","source":"estimators = list(best_estimators)","metadata":{"execution":{"iopub.status.busy":"2022-07-26T00:09:37.667899Z","iopub.execute_input":"2022-07-26T00:09:37.668351Z","iopub.status.idle":"2022-07-26T00:09:37.682232Z","shell.execute_reply.started":"2022-07-26T00:09:37.668320Z","shell.execute_reply":"2022-07-26T00:09:37.681292Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"estimators","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"execution":{"iopub.status.busy":"2022-07-26T00:09:37.683737Z","iopub.execute_input":"2022-07-26T00:09:37.684875Z","iopub.status.idle":"2022-07-26T00:09:37.703060Z","shell.execute_reply.started":"2022-07-26T00:09:37.684836Z","shell.execute_reply":"2022-07-26T00:09:37.702252Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Stacking Regressor ","metadata":{"execution":{"iopub.status.busy":"2022-06-29T03:26:28.279464Z","iopub.execute_input":"2022-06-29T03:26:28.279898Z","iopub.status.idle":"2022-06-29T03:26:28.285396Z","shell.execute_reply.started":"2022-06-29T03:26:28.279865Z","shell.execute_reply":"2022-06-29T03:26:28.284157Z"}}},{"cell_type":"code","source":"stack_overflow = StackingRegressor(estimators = estimators,cv=sk_split)","metadata":{"execution":{"iopub.status.busy":"2022-07-26T00:09:37.704558Z","iopub.execute_input":"2022-07-26T00:09:37.704980Z","iopub.status.idle":"2022-07-26T00:09:37.713383Z","shell.execute_reply.started":"2022-07-26T00:09:37.704948Z","shell.execute_reply":"2022-07-26T00:09:37.712571Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"stack_overflow.fit(X_train_prepped,y_train)","metadata":{"execution":{"iopub.status.busy":"2022-07-26T00:09:37.714472Z","iopub.execute_input":"2022-07-26T00:09:37.715461Z","iopub.status.idle":"2022-07-26T00:11:02.311830Z","shell.execute_reply.started":"2022-07-26T00:09:37.715430Z","shell.execute_reply":"2022-07-26T00:11:02.310453Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred = stack_overflow.predict(X_train_prepped)","metadata":{"execution":{"iopub.status.busy":"2022-07-26T00:11:02.313394Z","iopub.execute_input":"2022-07-26T00:11:02.316867Z","iopub.status.idle":"2022-07-26T00:11:04.036698Z","shell.execute_reply.started":"2022-07-26T00:11:02.316811Z","shell.execute_reply":"2022-07-26T00:11:04.035367Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred_test = stack_overflow.predict(X_test_prepped)","metadata":{"execution":{"iopub.status.busy":"2022-07-26T00:11:04.039036Z","iopub.execute_input":"2022-07-26T00:11:04.039998Z","iopub.status.idle":"2022-07-26T00:11:05.698125Z","shell.execute_reply.started":"2022-07-26T00:11:04.039948Z","shell.execute_reply":"2022-07-26T00:11:05.696880Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Assessing Accuracy of the Model \n\n* $R^2$ Score\n* $MSE$ (Mean Squared Error) Score\n* $RMSE$ (Root Mean Squared Error) Score\n* $MAE$ (Mean Absolute Error)Score","metadata":{}},{"cell_type":"markdown","source":"## $R^2$ Score\n\n$$R^2 = 1 - \\frac{\\sum\\limits_{i=1}^{N}(y_i - \\hat{y})^2}{\\sum\\limits_{i=1}^{N}(y_i - \\bar{y})^2}$$\n\n* $\\hat{y} = $predicted value of $y$\n* $\\bar{y} = $mean value of $y$","metadata":{}},{"cell_type":"code","source":"r2_score = stack_overflow.score(X_train_prepped,y_train)","metadata":{"execution":{"iopub.status.busy":"2022-07-26T00:11:05.700423Z","iopub.execute_input":"2022-07-26T00:11:05.701372Z","iopub.status.idle":"2022-07-26T00:11:07.402864Z","shell.execute_reply.started":"2022-07-26T00:11:05.701322Z","shell.execute_reply":"2022-07-26T00:11:07.401562Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## MSE & RMSE Score \n\n$$MSE = \\frac{1}{n} \\sum\\limits_{i=1}^{N}(y_i - \\hat{y})^2$$ $$RMSE = \\sqrt{MSE}$$\n* $\\hat{y} = $predicted value of $y$","metadata":{}},{"cell_type":"code","source":"MSE_score = MSE(y_train_reshaped,y_pred)","metadata":{"execution":{"iopub.status.busy":"2022-07-26T00:11:07.404801Z","iopub.execute_input":"2022-07-26T00:11:07.405597Z","iopub.status.idle":"2022-07-26T00:11:07.411207Z","shell.execute_reply.started":"2022-07-26T00:11:07.405557Z","shell.execute_reply":"2022-07-26T00:11:07.410093Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"RMSE_score = np.sqrt(MSE_score)","metadata":{"execution":{"iopub.status.busy":"2022-07-26T00:11:07.413075Z","iopub.execute_input":"2022-07-26T00:11:07.413843Z","iopub.status.idle":"2022-07-26T00:11:07.422839Z","shell.execute_reply.started":"2022-07-26T00:11:07.413801Z","shell.execute_reply":"2022-07-26T00:11:07.421640Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## MAE Score\n\n$$MAE = \\frac{1}{n} \\sum\\limits_{i=1}^{N}|y_i - \\hat{y}|$$\n* $\\hat{y} = $predicted value of $y$","metadata":{}},{"cell_type":"code","source":"MAE_score = MAE(y_train_reshaped,y_pred)","metadata":{"execution":{"iopub.status.busy":"2022-07-26T00:11:07.424817Z","iopub.execute_input":"2022-07-26T00:11:07.425673Z","iopub.status.idle":"2022-07-26T00:11:07.437198Z","shell.execute_reply.started":"2022-07-26T00:11:07.425628Z","shell.execute_reply":"2022-07-26T00:11:07.435644Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f'R2 Score:{r2_score}\\nMAE Score:{MAE_score}\\nMSE Score:{MSE_score}\\nRMSE Score:{RMSE_score}')","metadata":{"execution":{"iopub.status.busy":"2022-07-26T00:11:07.439458Z","iopub.execute_input":"2022-07-26T00:11:07.440331Z","iopub.status.idle":"2022-07-26T00:11:07.449518Z","shell.execute_reply.started":"2022-07-26T00:11:07.440289Z","shell.execute_reply":"2022-07-26T00:11:07.448264Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = px.scatter(x=y_train, y=y_pred, labels={'x': 'ground truth', 'y': 'prediction'},color_discrete_sequence=[\"dodgerblue\"])\nfig.add_shape(\n    type=\"line\", line=dict(dash='dash'),\n    x0=y_train.min(), y0=y_train.min(),\n    x1=y_train.max(), y1=y_train.max()\n) \nfig.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"residual = y_pred - y_train\nfig = px.scatter(x=y_pred, y=residual,\n    marginal_y='violin',trendline=\"ols\",labels={'x': 'Prediction', 'y': 'Residual'},color_discrete_sequence=[\"dodgerblue\"])\nfig.update_layout(title=\"Residual Plot\")\nfig.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Regression MLP\n\nFor good ol' fun.","metadata":{}},{"cell_type":"code","source":"import tensorflow as tf\nfrom tensorflow.keras.layers import Dense\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.callbacks import EarlyStopping","metadata":{"execution":{"iopub.status.busy":"2022-07-26T00:11:07.656016Z","iopub.execute_input":"2022-07-26T00:11:07.656342Z","iopub.status.idle":"2022-07-26T00:11:14.226861Z","shell.execute_reply.started":"2022-07-26T00:11:07.656312Z","shell.execute_reply":"2022-07-26T00:11:14.225604Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"n_cols = X_train_prepped.shape[1]","metadata":{"execution":{"iopub.status.busy":"2022-07-26T00:11:14.228201Z","iopub.execute_input":"2022-07-26T00:11:14.229457Z","iopub.status.idle":"2022-07-26T00:11:14.233499Z","shell.execute_reply.started":"2022-07-26T00:11:14.229419Z","shell.execute_reply":"2022-07-26T00:11:14.232790Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"loss_func = tf.keras.losses.MeanSquaredError()\neval_metrics = [tf.keras.metrics.RootMeanSquaredError(name=\"root_MSE\", dtype=None),\n                tf.keras.losses.MeanAbsoluteError(reduction=\"auto\", name=\"MAE\"),\n                tf.keras.metrics.MeanAbsolutePercentageError(name=\"mape\", dtype=None),\n                tf.keras.metrics.MeanSquaredLogarithmicError(name=\"msle\", dtype=None),\n                tf.keras.metrics.CosineSimilarity(name=\"cosine_similarity\", dtype=None, axis=-1),\n                tf.keras.metrics.LogCoshError(name=\"logcosh\", dtype=None),]","metadata":{"execution":{"iopub.status.busy":"2022-07-26T00:11:14.234494Z","iopub.execute_input":"2022-07-26T00:11:14.235014Z","iopub.status.idle":"2022-07-26T00:11:14.333639Z","shell.execute_reply.started":"2022-07-26T00:11:14.234983Z","shell.execute_reply":"2022-07-26T00:11:14.332662Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = Sequential()\nmodel.add(Dense(500,activation=\"relu\",input_shape=(n_cols,)))\n# model.add(Dense(500,activation=\"relu\"))\n# model.add(Dense(300,activation=\"relu\"))\n# model.add(Dense(200,activation=\"relu\"))\n# model.add(Dense(300,activation=\"relu\"))\n# model.add(Dense(300,activation=\"relu\"))\nmodel.add(Dense(1))\nmodel.compile(optimizer=\"adam\",\n              loss=loss_func,\n              metrics=eval_metrics)","metadata":{"execution":{"iopub.status.busy":"2022-07-26T00:11:14.335375Z","iopub.execute_input":"2022-07-26T00:11:14.335975Z","iopub.status.idle":"2022-07-26T00:11:14.689201Z","shell.execute_reply.started":"2022-07-26T00:11:14.335941Z","shell.execute_reply":"2022-07-26T00:11:14.688345Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.summary()","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2022-07-26T00:11:14.690398Z","iopub.execute_input":"2022-07-26T00:11:14.690972Z","iopub.status.idle":"2022-07-26T00:11:14.697583Z","shell.execute_reply.started":"2022-07-26T00:11:14.690939Z","shell.execute_reply":"2022-07-26T00:11:14.696511Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = model.fit(X_train_prepped,\n                    y_train,\n                    epochs=3000, # Just set it to a high enough number incase it doesn't early stop\n                    validation_split=0.3,\n                    verbose=not False,\n                   callbacks=[EarlyStopping(patience=20)]\n                   )","metadata":{"_kg_hide-output":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_neural_net_pred = model.predict(X_test_prepped)","metadata":{"execution":{"iopub.status.busy":"2022-07-26T00:11:25.511244Z","iopub.execute_input":"2022-07-26T00:11:25.511612Z","iopub.status.idle":"2022-07-26T00:11:25.771652Z","shell.execute_reply.started":"2022-07-26T00:11:25.511581Z","shell.execute_reply":"2022-07-26T00:11:25.770680Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## RMSE vs Number of Epochs","metadata":{}},{"cell_type":"code","source":"fig = go.Figure()\n\nlength = len(history.history[\"root_MSE\"])\nfig.add_trace(go.Scatter(x=list(range(1,length+1)), \n                         y=history.history[\"root_MSE\"],\n                         name=\"rmse\",mode=\"lines+markers\", \n                    line_shape='spline',line=dict(color=\"#AA02A0\")))\nfig.update_yaxes(title=\"Root Mean Squared Error\")\nfig.update_xaxes(title=\"# Epochs\")\nfig.update_layout(title=\"Root Mean Squared Error vs # Epochs\")","metadata":{"execution":{"iopub.status.busy":"2022-07-26T00:21:17.825470Z","iopub.execute_input":"2022-07-26T00:21:17.826188Z","iopub.status.idle":"2022-07-26T00:21:17.846067Z","shell.execute_reply.started":"2022-07-26T00:21:17.826132Z","shell.execute_reply":"2022-07-26T00:21:17.844769Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Mean Squared Logarithmic Error vs Number of Epochs","metadata":{}},{"cell_type":"code","source":"fig = go.Figure()\n\nlength = len(history.history[\"msle\"])\nfig.add_trace(go.Scatter(x=list(range(1,length+1)), \n                         y=history.history[\"msle\"],\n                         name=\"MSLE\", mode=\"lines+markers\",\n                    line_shape='spline',line=dict(color=\"#EA14E9\")))\nfig.update_yaxes(title=\"Mean Squared Logarithmic Error\")\nfig.update_xaxes(title=\"# Epochs\")\nfig.update_layout(title=\"Mean Squared Logarithmic Error vs # Epochs\")","metadata":{"execution":{"iopub.status.busy":"2022-07-26T00:18:25.661268Z","iopub.execute_input":"2022-07-26T00:18:25.661649Z","iopub.status.idle":"2022-07-26T00:18:25.683954Z","shell.execute_reply.started":"2022-07-26T00:18:25.661618Z","shell.execute_reply":"2022-07-26T00:18:25.682620Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## MAE vs Number of Epochs","metadata":{}},{"cell_type":"code","source":"fig = go.Figure()\n\nlength = len(history.history[\"MAE\"])\nfig.add_trace(go.Scatter(x=list(range(1,length+1)), \n                         y=history.history[\"MAE\"],\n                         name=\"mae\", mode=\"lines+markers\",\n                    line_shape='spline',line=dict(color=\"#1E80F9\")))\nfig.update_yaxes(title=\"Mean Absolute Error\")\nfig.update_xaxes(title=\"# Epochs\")\nfig.update_layout(title=\"Mean Absolute Error vs # Epochs\")","metadata":{"execution":{"iopub.status.busy":"2022-07-26T00:20:49.481952Z","iopub.execute_input":"2022-07-26T00:20:49.482346Z","iopub.status.idle":"2022-07-26T00:20:49.506376Z","shell.execute_reply.started":"2022-07-26T00:20:49.482316Z","shell.execute_reply":"2022-07-26T00:20:49.505375Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Mean Absolute Percentage Error vs # Epochs","metadata":{}},{"cell_type":"code","source":"fig = go.Figure()\n\nlength = len(history.history[\"mape\"])\nfig.add_trace(go.Scatter(x=list(range(1,length+1)), \n                         y=history.history[\"mape\"],\n                         name=\"mape\",mode=\"lines+markers\",\n                    line_shape='spline',line=dict(color=\"#0D79F1\")))\nfig.update_yaxes(title=\"Mean Absolute Percentage Error\")\nfig.update_xaxes(title=\"# Epochs\")\nfig.update_layout(title=\"Mean Absolute Percentage Error vs # Epochs\")","metadata":{"execution":{"iopub.status.busy":"2022-07-26T00:21:48.290849Z","iopub.execute_input":"2022-07-26T00:21:48.291258Z","iopub.status.idle":"2022-07-26T00:21:48.310737Z","shell.execute_reply.started":"2022-07-26T00:21:48.291225Z","shell.execute_reply":"2022-07-26T00:21:48.309533Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Cosine Similarity vs # Epochs","metadata":{}},{"cell_type":"code","source":"fig = go.Figure()\n\nlength = len(history.history[\"cosine_similarity\"])\nfig.add_trace(go.Scatter(x=list(range(1,length+1)), \n                         y=history.history[\"cosine_similarity\"],\n                         name=\"Cosine Similarity\", mode=\"lines+markers\",\n                    line_shape='spline',line=dict(color=\"rgb(207, 64, 159)\")))\nfig.update_yaxes(title=\"Cosine Similarity\")\nfig.update_xaxes(title=\"# Epochs\")\nfig.update_layout(title=\"Cosine Similarity vs # Epochs\")","metadata":{"execution":{"iopub.status.busy":"2022-07-26T00:22:00.591166Z","iopub.execute_input":"2022-07-26T00:22:00.591588Z","iopub.status.idle":"2022-07-26T00:22:00.614841Z","shell.execute_reply.started":"2022-07-26T00:22:00.591553Z","shell.execute_reply":"2022-07-26T00:22:00.613889Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Log-Cosh vs Number of Epochs","metadata":{}},{"cell_type":"code","source":"fig = go.Figure()\n\nlength = len(history.history[\"logcosh\"])\nfig.add_trace(go.Scatter(x=list(range(1,length+1)), \n                         y=history.history[\"logcosh\"],\n                         name=\"log_cosh\", mode=\"lines+markers\",\n                    line_shape='spline',line=dict(color='rgb(162, 83, 40)')))\nfig.update_yaxes(title=\"Log Cosh\")\nfig.update_xaxes(title=\"# Epochs\")\nfig.update_layout(title=\"Log Cosh vs # Epochs\")","metadata":{"execution":{"iopub.status.busy":"2022-07-26T00:20:07.342111Z","iopub.execute_input":"2022-07-26T00:20:07.342635Z","iopub.status.idle":"2022-07-26T00:20:07.370919Z","shell.execute_reply.started":"2022-07-26T00:20:07.342594Z","shell.execute_reply":"2022-07-26T00:20:07.369766Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Submission","metadata":{}},{"cell_type":"code","source":"output = pd.DataFrame(columns=[\"Id\",\"SalePrice\"])\noutput[\"Id\"] = test_id_cols\noutput[\"SalePrice\"] = list(y_neural_net_pred.flatten())\noutput[\"Id\"] = output[\"Id\"].astype(int)","metadata":{"execution":{"iopub.status.busy":"2022-07-26T00:17:04.968069Z","iopub.execute_input":"2022-07-26T00:17:04.968475Z","iopub.status.idle":"2022-07-26T00:17:04.980050Z","shell.execute_reply.started":"2022-07-26T00:17:04.968441Z","shell.execute_reply":"2022-07-26T00:17:04.979240Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"output.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-26T00:11:25.923795Z","iopub.execute_input":"2022-07-26T00:11:25.924145Z","iopub.status.idle":"2022-07-26T00:11:25.939335Z","shell.execute_reply.started":"2022-07-26T00:11:25.924114Z","shell.execute_reply":"2022-07-26T00:11:25.938196Z"},"trusted":true},"execution_count":null,"outputs":[]}]}