{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Chapter: Preparation","metadata":{"papermill":{"duration":0.021022,"end_time":"2022-07-28T12:31:26.926174","exception":false,"start_time":"2022-07-28T12:31:26.905152","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"## Set parameters for platform operation\nEnvironmental adjustments to absorb platform differences","metadata":{"papermill":{"duration":0.020907,"end_time":"2022-07-28T12:31:26.968066","exception":false,"start_time":"2022-07-28T12:31:26.947159","status":"completed"},"tags":[]}},{"cell_type":"code","source":"EXTERNAL       = False                             # kaggle-> False, External platforms -> True\nPROJECT_NAME   = \"home-data-for-ml-course\"         # competition project name.\n\nif EXTERNAL:\n    print(\"Set the 'kaggle.json' at working directory before running subsequent cells.\")\nelse:\n    print(\"Excuting this notebook on Kaggle platform.\")","metadata":{"papermill":{"duration":0.040067,"end_time":"2022-07-28T12:31:27.027915","exception":false,"start_time":"2022-07-28T12:31:26.987848","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-29T09:54:15.809003Z","iopub.execute_input":"2022-07-29T09:54:15.809489Z","iopub.status.idle":"2022-07-29T09:54:15.849306Z","shell.execute_reply.started":"2022-07-29T09:54:15.809377Z","shell.execute_reply":"2022-07-29T09:54:15.847590Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Additional processing on platforms other than Kaggle\n* Setting up kaggle.json for using Kaggle API.\n* Installing libraries needed to run this notebook, those is preinstalled on Kaggle.(Those depends on the environment)","metadata":{"papermill":{"duration":0.018796,"end_time":"2022-07-28T12:31:27.067417","exception":false,"start_time":"2022-07-28T12:31:27.048621","status":"completed"},"tags":[]}},{"cell_type":"code","source":"%%bash -s $EXTERNAL $PROJECT_NAME\nif [ $1 = \"True\" ]; then\n    echo \"external mode.\"\n\n    # Create directories.\n    mkdir -p ~/.kaggle\n    mkdir -p input\n    mkdir -p working\n    \n    # Put kaggle.json in place.\n    cp kaggle.json ~/.kaggle/\n    chmod 600 ~/.kaggle/kaggle.json\n\n    # Download and extract zip file.\n    pip install -q kaggle\n    kaggle competitions download -c $2 -p input\n    unzip input/$2.zip -d input/$2\n\n\n    # Install libraries\n    pip install -q category_encoders\n    pip install -q optuna\n\n    echo \"END\"\nelse\n    echo \"not external mode.\"\nfi","metadata":{"papermill":{"duration":0.048447,"end_time":"2022-07-28T12:31:27.134926","exception":false,"start_time":"2022-07-28T12:31:27.086479","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-29T09:54:25.407371Z","iopub.execute_input":"2022-07-29T09:54:25.407860Z","iopub.status.idle":"2022-07-29T09:54:25.431421Z","shell.execute_reply.started":"2022-07-29T09:54:25.407826Z","shell.execute_reply":"2022-07-29T09:54:25.430123Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if EXTERNAL:\n    %cd working\n!pwd","metadata":{"papermill":{"duration":0.779542,"end_time":"2022-07-28T12:31:27.934557","exception":false,"start_time":"2022-07-28T12:31:27.155015","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-29T09:54:26.667425Z","iopub.execute_input":"2022-07-29T09:54:26.667840Z","iopub.status.idle":"2022-07-29T09:54:27.439296Z","shell.execute_reply.started":"2022-07-29T09:54:26.667806Z","shell.execute_reply":"2022-07-29T09:54:27.437704Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Import libraries","metadata":{"papermill":{"duration":0.019151,"end_time":"2022-07-28T12:31:27.973146","exception":false,"start_time":"2022-07-28T12:31:27.953995","status":"completed"},"tags":[]}},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns\nimport random\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.compose import ColumnTransformer\nfrom sklearn.preprocessing import FunctionTransformer\nfrom sklearn.preprocessing import OneHotEncoder\nfrom sklearn.preprocessing import OrdinalEncoder\nimport category_encoders as ce\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.decomposition import PCA\nfrom sklearn.base import BaseEstimator, TransformerMixin\nfrom sklearn.feature_selection import RFECV\nfrom sklearn.feature_selection import SelectFromModel\nfrom sklearn.feature_selection import SequentialFeatureSelector\nfrom sklearn.ensemble import RandomForestRegressor\nimport optuna\nfrom xgboost import XGBRegressor\n\nfrom sklearn.model_selection import KFold\nfrom sklearn.model_selection import cross_val_score\nfrom sklearn.model_selection import GridSearchCV\nfrom sklearn import set_config\n\n!pip install -q missingno\nimport missingno as msno\n\nfrom sklearn.ensemble import GradientBoostingRegressor\nfrom sklearn.ensemble import ExtraTreesRegressor\nfrom sklearn.neighbors import KNeighborsRegressor\n\nimport sys\nfrom sklearn.model_selection import KFold\nfrom tqdm import tqdm\n\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom collections import defaultdict\nfrom lightgbm import LGBMRegressor\nimport math\nfrom sklearn.cluster import KMeans\nfrom sklearn.neighbors import LocalOutlierFactor\n","metadata":{"papermill":{"duration":16.224229,"end_time":"2022-07-28T12:31:44.216812","exception":false,"start_time":"2022-07-28T12:31:27.992583","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-29T09:54:29.483456Z","iopub.execute_input":"2022-07-29T09:54:29.483919Z","iopub.status.idle":"2022-07-29T09:54:46.897813Z","shell.execute_reply.started":"2022-07-29T09:54:29.483879Z","shell.execute_reply":"2022-07-29T09:54:46.896456Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('../input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","papermill":{"duration":0.033499,"end_time":"2022-07-28T12:31:44.269807","exception":false,"start_time":"2022-07-28T12:31:44.236308","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-29T09:54:46.899925Z","iopub.execute_input":"2022-07-29T09:54:46.900322Z","iopub.status.idle":"2022-07-29T09:54:46.910545Z","shell.execute_reply.started":"2022-07-29T09:54:46.900286Z","shell.execute_reply":"2022-07-29T09:54:46.909269Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Chapter: Data Analysis","metadata":{"papermill":{"duration":0.019139,"end_time":"2022-07-28T12:31:44.308621","exception":false,"start_time":"2022-07-28T12:31:44.289482","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"## Load and preliminary exploring the train dataset¶","metadata":{"papermill":{"duration":0.019844,"end_time":"2022-07-28T12:31:44.347892","exception":false,"start_time":"2022-07-28T12:31:44.328048","status":"completed"},"tags":[]}},{"cell_type":"code","source":"def load_csv(datafile_path,contains_target_column=True):\n    X_full = pd.read_csv(datafile_path,index_col=\"Id\")\n    if contains_target_column:\n        y = X_full.SalePrice\n        X = X_full.drop([\"SalePrice\"],axis=1)\n        return X_full,X,y\n    else:\n        return X_full\n","metadata":{"papermill":{"duration":0.029241,"end_time":"2022-07-28T12:31:44.396670","exception":false,"start_time":"2022-07-28T12:31:44.367429","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-29T09:56:23.783021Z","iopub.execute_input":"2022-07-29T09:56:23.783477Z","iopub.status.idle":"2022-07-29T09:56:23.791775Z","shell.execute_reply.started":"2022-07-29T09:56:23.783441Z","shell.execute_reply":"2022-07-29T09:56:23.790348Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_full,train_X,train_y = load_csv(\"../input/home-data-for-ml-course/train.csv\",contains_target_column=True)\n\nprint(\"### shape ##########\")\nprint(\"train_X.shape=\",train_X.shape)\nprint(\"train_y.shape=\",train_y.shape)\nprint()\nprint(\"### X_full head ##########\")\ndisplay(X_full.head())\nprint()\nprint(\"### X_full describing #########\")\ndisplay(X_full.describe())\nprint()\nprint(\"### X_full info #########\")\nX_full.info()\n","metadata":{"papermill":{"duration":0.237043,"end_time":"2022-07-28T12:31:44.653240","exception":false,"start_time":"2022-07-28T12:31:44.416197","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-29T09:56:24.861854Z","iopub.execute_input":"2022-07-29T09:56:24.863410Z","iopub.status.idle":"2022-07-29T09:56:25.126607Z","shell.execute_reply.started":"2022-07-29T09:56:24.863334Z","shell.execute_reply":"2022-07-29T09:56:25.125619Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Missing values","metadata":{"papermill":{"duration":0.020591,"end_time":"2022-07-28T12:31:44.694854","exception":false,"start_time":"2022-07-28T12:31:44.674263","status":"completed"},"tags":[]}},{"cell_type":"code","source":"fig,ax=plt.subplots(figsize=(25,10))\np=sns.heatmap(X_full.isnull(), cbar=False)\nplt.title(\"Missing pattern of features (X_full).\")\nplt.show()\n","metadata":{"papermill":{"duration":1.954182,"end_time":"2022-07-28T12:31:46.669930","exception":false,"start_time":"2022-07-28T12:31:44.715748","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-29T09:56:26.301586Z","iopub.execute_input":"2022-07-29T09:56:26.302019Z","iopub.status.idle":"2022-07-29T09:56:28.505898Z","shell.execute_reply.started":"2022-07-29T09:56:26.301984Z","shell.execute_reply":"2022-07-29T09:56:28.504866Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"### train_X.isnull().sum()>0  ##########\")\ntmp1 = train_X.isnull().sum()\ndisplay(tmp1[tmp1>0])\nprint()\nprint(\"### train_X missing rates( over 0 )    ##########\")\ntmp2 = train_X.isnull().sum()/train_X.shape[0]\ndisplay(tmp2[tmp2>0])\nprint()\nprint(\"### train_X hight missing rates( over 0.1).  ##########\")\ndisplay(tmp2[tmp2>0.1])\nprint()\nprint(\"#######################\")\nprint(\"high missing rate cols =\",list(tmp2[tmp2>0.1].index))","metadata":{"papermill":{"duration":0.067922,"end_time":"2022-07-28T12:31:46.760473","exception":false,"start_time":"2022-07-28T12:31:46.692551","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-29T09:56:28.507650Z","iopub.execute_input":"2022-07-29T09:56:28.508543Z","iopub.status.idle":"2022-07-29T09:56:28.558824Z","shell.execute_reply.started":"2022-07-29T09:56:28.508505Z","shell.execute_reply":"2022-07-29T09:56:28.557471Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### note\n* high missin rate cols, 'LotFrontage', 'Alley', 'FireplaceQu', 'PoolQC', 'Fence' and 'MiscFeature' aren't suitable for inputting models.","metadata":{"papermill":{"duration":0.02276,"end_time":"2022-07-28T12:31:46.806573","exception":false,"start_time":"2022-07-28T12:31:46.783813","status":"completed"},"tags":[]}},{"cell_type":"code","source":"msno.heatmap(X_full)\nplt.title(\"Correlations of missing features.\")\nplt.show()","metadata":{"papermill":{"duration":0.939215,"end_time":"2022-07-28T12:31:47.768894","exception":false,"start_time":"2022-07-28T12:31:46.829679","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-29T09:56:30.078791Z","iopub.execute_input":"2022-07-29T09:56:30.079998Z","iopub.status.idle":"2022-07-29T09:56:31.100430Z","shell.execute_reply.started":"2022-07-29T09:56:30.079922Z","shell.execute_reply":"2022-07-29T09:56:31.098295Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Cardinality","metadata":{"papermill":{"duration":0.024741,"end_time":"2022-07-28T12:31:47.819459","exception":false,"start_time":"2022-07-28T12:31:47.794718","status":"completed"},"tags":[]}},{"cell_type":"code","source":"tmp=X_full.select_dtypes(\"object\").nunique()\ndisplay(tmp.sort_values(ascending=False))\nprint()\nprint(\"high_cardinality_cols=\",tmp[tmp>=10].index)","metadata":{"papermill":{"duration":0.053412,"end_time":"2022-07-28T12:31:47.897875","exception":false,"start_time":"2022-07-28T12:31:47.844463","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-29T09:56:31.456155Z","iopub.execute_input":"2022-07-29T09:56:31.456998Z","iopub.status.idle":"2022-07-29T09:56:31.487286Z","shell.execute_reply.started":"2022-07-29T09:56:31.456957Z","shell.execute_reply":"2022-07-29T09:56:31.485928Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### note\n* It is necessary to use category encoder algorithms properly according to the level of cardinality.","metadata":{"papermill":{"duration":0.025091,"end_time":"2022-07-28T12:31:47.947567","exception":false,"start_time":"2022-07-28T12:31:47.922476","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"## Roughly overlook the objective distribution in train dataset\nFor reference, outliers are classified by color (contamination rate is 5%).","metadata":{"papermill":{"duration":0.024406,"end_time":"2022-07-28T12:31:47.996708","exception":false,"start_time":"2022-07-28T12:31:47.972302","status":"completed"},"tags":[]}},{"cell_type":"code","source":"CONTAMINATION_RATE=0.05\n\n_,_,train_y = load_csv(\"../input/home-data-for-ml-course/train.csv\",contains_target_column=True)\nprint(\"train_y.shape=\",train_y.shape)\n\ntrain_outlier_labels = LocalOutlierFactor(contamination=CONTAMINATION_RATE).fit_predict(train_y.to_frame())\ntrain_outlier_labels = train_outlier_labels ==-1\ntmp_train_inlier = train_y.to_frame()[~train_outlier_labels]\ntmp_train_outlier = train_y.to_frame()[train_outlier_labels]\nprint(\"contamination rate:\",CONTAMINATION_RATE)\nprint(\"Number of outlier samples:\",train_outlier_labels.sum())\n\nbins = np.histogram(train_y,bins=50)[1]\nfig,axes =plt.subplots(2,1,figsize=(20,10),constrained_layout=True)\nsns.histplot(data=tmp_train_inlier,x=\"SalePrice\",label=\"inlier\",color=\"teal\",ax=axes[0],bins=bins,edgecolor=\"darkslategray\",linewidth=2)\nsns.histplot(data=tmp_train_outlier,x=\"SalePrice\",label=f\"outlier {CONTAMINATION_RATE:.1%}\",color=\"magenta\",ax=axes[0],bins=bins,alpha=0.5,edgecolor=\"darkmagenta\",linewidth=2)\nsns.stripplot(data=tmp_train_inlier,x=\"SalePrice\",label=\"inlier\",color=\"teal\",ax=axes[1])\nsns.stripplot(data=tmp_train_outlier,x=\"SalePrice\",label=f\"outlier {CONTAMINATION_RATE:.1%}\",color=\"darkmagenta\",ax=axes[1])\naxes[0].legend()\naxes[1].legend()\nplt.show()\n","metadata":{"papermill":{"duration":0.865466,"end_time":"2022-07-28T12:31:48.887155","exception":false,"start_time":"2022-07-28T12:31:48.021689","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-29T09:56:33.841040Z","iopub.execute_input":"2022-07-29T09:56:33.841467Z","iopub.status.idle":"2022-07-29T09:56:34.764111Z","shell.execute_reply.started":"2022-07-29T09:56:33.841434Z","shell.execute_reply":"2022-07-29T09:56:34.762487Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Correlations of quantitative features","metadata":{"papermill":{"duration":0.026208,"end_time":"2022-07-28T12:31:48.941046","exception":false,"start_time":"2022-07-28T12:31:48.914838","status":"completed"},"tags":[]}},{"cell_type":"code","source":"# Sorted and displayed in descending order of correlation with the objective variable.\nX_full_corr = X_full.corr()\nX_full_corr = X_full_corr.sort_values(\"SalePrice\",ascending=False)\nX_full_corr = X_full_corr[X_full_corr.index]\nX_full_corr.style.background_gradient(axis=None,cmap=\"bwr\",vmin=-1,vmax=1)","metadata":{"papermill":{"duration":0.236058,"end_time":"2022-07-28T12:31:49.203743","exception":false,"start_time":"2022-07-28T12:31:48.967685","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-29T09:56:36.435253Z","iopub.execute_input":"2022-07-29T09:56:36.436978Z","iopub.status.idle":"2022-07-29T09:56:36.658652Z","shell.execute_reply.started":"2022-07-29T09:56:36.436914Z","shell.execute_reply":"2022-07-29T09:56:36.657035Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Relations between objective and features","metadata":{"papermill":{"duration":0.032311,"end_time":"2022-07-28T12:31:49.267697","exception":false,"start_time":"2022-07-28T12:31:49.235386","status":"completed"},"tags":[]}},{"cell_type":"code","source":"N_COLS=5\nn_rows = math.ceil(len(X_full.columns)/N_COLS)\n\nfig,axes =plt.subplots(n_rows,N_COLS,figsize=(30,50),constrained_layout=True)\nfor col,ax in zip(X_full.columns,axes.ravel()):\n    if X_full[col].dtypes ==\"object\":\n        sns.boxplot(ax=ax,x=col,y=\"SalePrice\",data=X_full)\n    else:\n        sns.scatterplot(ax=ax,x=col,y=\"SalePrice\",data=X_full)\n\nplt.show()","metadata":{"papermill":{"duration":17.411888,"end_time":"2022-07-28T12:32:06.710536","exception":false,"start_time":"2022-07-28T12:31:49.298648","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-29T09:56:39.330576Z","iopub.execute_input":"2022-07-29T09:56:39.331651Z","iopub.status.idle":"2022-07-29T09:56:58.750037Z","shell.execute_reply.started":"2022-07-29T09:56:39.331582Z","shell.execute_reply":"2022-07-29T09:56:58.749048Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Comparison of features' distribution between train and test data","metadata":{"papermill":{"duration":0.047435,"end_time":"2022-07-28T12:32:06.805752","exception":false,"start_time":"2022-07-28T12:32:06.758317","status":"completed"},"tags":[]}},{"cell_type":"code","source":"train_X_full,train_X,_ = load_csv(\"../input/home-data-for-ml-course/train.csv\",contains_target_column=True)\ntest_X                 = load_csv(\"../input/home-data-for-ml-course/test.csv\",contains_target_column=False)\nprint(\"train_X.shape=\",train_X.shape)\nprint(\"test_X.shape=\",test_X.shape)\nprint(\"Matching check of feature sequence:\",(~(train_X.columns==test_X.columns)).sum()==0)\n\nN_COLS=3\n\nn_fig_rows = math.ceil(len(train_X.columns)/N_COLS)\nfig,axes =plt.subplots(n_fig_rows,N_COLS,figsize=(10*N_COLS,5*n_fig_rows),constrained_layout=True)\ncol_no,row_no = 0,0\nfor ft_no,ft_name in enumerate(train_X.columns):\n    row_no = ft_no // N_COLS\n    col_no = ft_no % N_COLS\n    if train_X[ft_name].dtype in [\"object\",\"category\"]:\n        x_order=pd.concat([train_X[ft_name],test_X[ft_name]],axis=0).unique()\n        sns.countplot(data=train_X,x=ft_name,ax=axes[row_no,col_no],order=x_order,label=\"train\",color=\"teal\",edgecolor=\"teal\",linewidth=2)\n        sns.countplot(data=test_X, x=ft_name,ax=axes[row_no,col_no],order=x_order,label=\"test\", color=\"magenta\",alpha=0.5,edgecolor=\"darkviolet\",linewidth=2)\n    else:\n        bins=np.histogram(pd.concat([train_X[ft_name],test_X[ft_name]],axis=0).dropna(), bins=50)[1]\n        sns.histplot(train_X[ft_name],ax=axes[row_no,col_no],bins=bins,label=\"train\",color=\"teal\",edgecolor=\"teal\",linewidth=2)\n        sns.histplot(test_X[ft_name] ,ax=axes[row_no,col_no],bins=bins,label=\"test\", color=\"magenta\",alpha=0.5,edgecolor=\"darkviolet\",linewidth=2)\n    axes[row_no,col_no].set_title(ft_name,loc=\"center\",x=0.5,y=0.9,fontsize=20)\n    axes[row_no,col_no].legend(loc=\"upper right\",fontsize=20)\n    axes[row_no,col_no].set_xlabel(None)\n    \nplt.show()\n","metadata":{"papermill":{"duration":25.585568,"end_time":"2022-07-28T12:32:32.438116","exception":false,"start_time":"2022-07-28T12:32:06.852548","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-29T09:56:58.751754Z","iopub.execute_input":"2022-07-29T09:56:58.752611Z","iopub.status.idle":"2022-07-29T09:57:26.350968Z","shell.execute_reply.started":"2022-07-29T09:56:58.752566Z","shell.execute_reply":"2022-07-29T09:57:26.349773Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### note\n* There is almost no difference in the distribution of objective variables between the training and the test dataset.\n* There seems to be no particular factor that influences inference.","metadata":{"papermill":{"duration":0.060093,"end_time":"2022-07-28T12:32:32.560723","exception":false,"start_time":"2022-07-28T12:32:32.500630","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"## classify features, and identifying features that should be excluded.","metadata":{"papermill":{"duration":0.064747,"end_time":"2022-07-28T12:32:32.699554","exception":false,"start_time":"2022-07-28T12:32:32.634807","status":"completed"},"tags":[]}},{"cell_type":"code","source":"def subtract_cols(input_cols,exclude_cols):\n    return list(set(input_cols)-set(exclude_cols))\n\n\n# classify features into categorical and numerical.\ncategorical_cols   = [col for col in train_X.columns if train_X[col].dtype == \"object\"]\nnumerical_cols  = [col for col in train_X.columns if train_X[col].dtype in [\"int64\",\"float64\"]]\n\n\n# Move features, based on description note.\nmove_to_categorical_cols=[\"MSSubClass\"]\n#move_to_categorical_cols=[\"MSSubClass\",\"OverallQual\",\"OverallCond\",\"MoSold\",\"YrSold\",\"YearBuilt\",\"GarageYrBlt\",\"YearRemodAdd\"]\ncategorical_cols  = categorical_cols + move_to_categorical_cols\nnumerical_cols    = subtract_cols(numerical_cols,move_to_categorical_cols)\n\n\n# Exclude sale data for preventing from target leaking.\ntarget_leaking_cols = ['YrSold','SaleType','SaleCondition','MoSold']\nnumerical_cols    = subtract_cols(numerical_cols  , target_leaking_cols)\ncategorical_cols  = subtract_cols(categorical_cols, target_leaking_cols)\n\n\n\n\n# exclude high rate missing features.\ntmp = train_X.isnull().sum()/train_X.shape[0]\nhigh_rate_missing_cols = list(tmp[tmp>0.1].index)\nnumerical_cols    = subtract_cols(numerical_cols  ,high_rate_missing_cols)\ncategorical_cols  = subtract_cols(categorical_cols,high_rate_missing_cols)\n\n\n# classify object features into high and low cardinality.\nlow_cardinality_cols  = [col for col in categorical_cols if train_X[col].nunique()<10]\nhigh_cardinality_cols = subtract_cols(categorical_cols, low_cardinality_cols)\n\n\n# Features list summary¶\nprint(\"low_cardinality_cols:\",low_cardinality_cols)\nprint()\nprint(\"high_cardinality_cols:\",high_cardinality_cols)\nprint()\nprint(\"numerical_cols:\",numerical_cols)\n","metadata":{"papermill":{"duration":0.094486,"end_time":"2022-07-28T12:32:32.854326","exception":false,"start_time":"2022-07-28T12:32:32.759840","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-29T09:57:26.352798Z","iopub.execute_input":"2022-07-29T09:57:26.353372Z","iopub.status.idle":"2022-07-29T09:57:26.386876Z","shell.execute_reply.started":"2022-07-29T09:57:26.353339Z","shell.execute_reply":"2022-07-29T09:57:26.385882Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Create features additinonaly","metadata":{"papermill":{"duration":0.058658,"end_time":"2022-07-28T12:32:32.972560","exception":false,"start_time":"2022-07-28T12:32:32.913902","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"Create features, following the teaching of the kaggle tutorial course \"Feature Engineering\".\n1. Ratio of residential area to site area on the ground\n2. Area per room (using the number of rooms excluding basement and bathroom)\n3. External living area (total area of wooden deck, open pouch, fenced pouch, 3-season pouch, screen pouch)\n4. Number of types of outdoor living space\n5. Housing area for each housing type (excluding underground living area) *The implementation is being pended.\n6. Clustering the set of several features\n","metadata":{"papermill":{"duration":0.058687,"end_time":"2022-07-28T12:32:33.090617","exception":false,"start_time":"2022-07-28T12:32:33.031930","status":"completed"},"tags":[]}},{"cell_type":"code","source":"print(\"### 1.Ratio of residential area to site area on the ground ###################\")\ntmp_X = train_X.copy()[[\"GrLivArea\",\"LotArea\"]]\ntmp_X[\"LivLotRatio\"] = tmp_X[\"GrLivArea\"]/tmp_X[\"LotArea\"]\ndisplay(tmp_X.head().style.applymap(lambda x:\"background-color:skyblue\", subset=[\"LivLotRatio\"]))\nprint()\nprint()\n\nprint(\"### 2.Area per room (using the number of rooms excluding basement and bathroom) ###############\")\ntmp_X = train_X.copy()[[\"1stFlrSF\",\"2ndFlrSF\",\"TotRmsAbvGrd\"]]\ntmp_X[\"Spaciousness\"] = (tmp_X[\"1stFlrSF\"]+tmp_X[\"2ndFlrSF\"]) / tmp_X[\"TotRmsAbvGrd\"]\ndisplay(tmp_X.head().style.applymap(lambda x:\"background-color:skyblue\", subset=[\"Spaciousness\"]))\nprint()\nprint()\n\nprint(\"### 3.External living area (total area of wooden deck, open pouch, fenced pouch, 3-season pouch, screen pouch) ############\")\ntmp_X = train_X.copy()[[\"WoodDeckSF\",\"OpenPorchSF\",\"EnclosedPorch\",\"3SsnPorch\",\"ScreenPorch\"]]\ntmp_X[\"TotalOutsideSF\"] = tmp_X[[\"WoodDeckSF\",\"OpenPorchSF\",\"EnclosedPorch\",\"3SsnPorch\",\"ScreenPorch\"]].sum(axis=1)\ndisplay(tmp_X.head().style.applymap(lambda x:\"background-color:skyblue\", subset=[\"TotalOutsideSF\"]))\nprint()\nprint()\n\nprint(\"### 4.Number of types of outdoor living space ############\")\ntmp_X = train_X.copy()[[\"WoodDeckSF\",\"OpenPorchSF\",\"EnclosedPorch\",\"3SsnPorch\",\"ScreenPorch\"]]\ntmp_X[\"PorchTypes\"] = tmp_X.gt(0.0).sum(axis=1)\ndisplay(tmp_X.head().style.applymap(lambda x:\"background-color:skyblue\", subset=[\"PorchTypes\"]))\nprint()\nprint()\n\nprint(\"### 5.Housing area for each housing type (excluding underground living area ######################\")\ntmp_X = train_X.copy()[[\"BldgType\",\"GrLivArea\"]]\ntmp_out = pd.get_dummies(train_X[\"BldgType\"], prefix=\"Bldg\")\ntmp_out = tmp_out.mul(tmp_X[\"GrLivArea\"],axis=0)\ntmp_X =pd.concat([tmp_X,tmp_out],axis=1)\ndisplay(tmp_X.head().style.applymap(lambda x:\"background-color:skyblue\", subset=tmp_X.filter(like='Bldg_',axis=1).columns))\nprint(\"******************************************************\")\nprint(\"******** The implementation is being pended. *********\")\nprint(\"******************************************************\")\nprint()\nprint()\n\n\n\nprint(\"### 6. Clustering the set of several features ############\")\ntmp_X = train_X.copy()[[\"LotArea\", \"TotalBsmtSF\", \"1stFlrSF\", \"2ndFlrSF\",\"GrLivArea\"]]\nkmeans = KMeans(n_clusters=10, n_init=10, random_state=0)\nX_scaled = (tmp_X - tmp_X.mean(axis=0)) / tmp_X.std(axis=0)\ntmp_X[[f\"Cluster{idx}\" for idx in range(10)]] = kmeans.fit_transform(X_scaled)\ndisplay(tmp_X.head().style.applymap(lambda x:\"background-color:skyblue\", subset=tmp_X.filter(like='Cluster',axis=1).columns))\n\n","metadata":{"papermill":{"duration":0.309147,"end_time":"2022-07-28T12:32:33.458608","exception":false,"start_time":"2022-07-28T12:32:33.149461","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-29T09:57:26.389167Z","iopub.execute_input":"2022-07-29T09:57:26.389747Z","iopub.status.idle":"2022-07-29T09:57:26.674539Z","shell.execute_reply.started":"2022-07-29T09:57:26.389709Z","shell.execute_reply":"2022-07-29T09:57:26.672294Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Chapter: Building Pipelines","metadata":{"papermill":{"duration":0.065341,"end_time":"2022-07-28T12:32:33.584034","exception":false,"start_time":"2022-07-28T12:32:33.518693","status":"completed"},"tags":[]}},{"cell_type":"code","source":"from copy import deepcopy\n\ndef make_pipline_set(encoder_desc,lc_encoder_class,lc_encoder_params,hc_encoder_class,hc_encoder_params):\n    \n    \n    # ---Build simple pipeline ---------------------------------------------------------------------\n    numerical_pipe = Pipeline(steps=[\n        (\"impute\",SimpleImputer(strategy=\"median\")),\n        (\"scale\", StandardScaler())\n    ])\n\n    categorical_lc_pipe = Pipeline(steps=[\n        (\"impute\", SimpleImputer(strategy=\"most_frequent\")),\n        (\"stringify\", FunctionTransformer(lambda aaa: aaa.astype('object'))),  # category_encoders needs \"object\" as target.\n        (\"low cardinality encoder\",lc_encoder_class(**lc_encoder_params))\n    ])\n    categorical_hc_pipe = Pipeline(steps=[\n        (\"impute\", SimpleImputer(strategy=\"most_frequent\")),\n        (\"stringify\", FunctionTransformer(lambda aaa: aaa.astype('object'))),  # category_encoders needs \"object\" as target.\n        (\"high cardinality encoder\",hc_encoder_class(**hc_encoder_params))\n    ])\n    \n    # total pipeline\n    features_pipeline = ColumnTransformer(transformers=[\n        (\"numerical\",      numerical_pipe,     numerical_cols),\n        (\"categorical_lc\", categorical_lc_pipe,   low_cardinality_cols),\n        (\"categorical_hc\", categorical_hc_pipe,   high_cardinality_cols),\n    ])\n\n    \n    # ---Build pipeline with creating features---------------------------------------------------------------------\n    class CreateFeaturesPipeline(BaseEstimator, TransformerMixin):\n        def __init__(self):\n            super().__init__()\n\n        def fit(self, X, y=None):\n            return self\n\n        def transform(self, X):\n            # 1. Ratio of residential area to site area on the ground\n            tmp = X[[\"GrLivArea\",\"LotArea\"]].copy()\n            tmp[\"LivLotRatio\"] = tmp[\"GrLivArea\"] / tmp[\"LotArea\"]\n            df_out=tmp[[\"LivLotRatio\"]]\n\n            # 2. Area per room (using the number of rooms excluding basement and bathroom)\n            tmp = X[[\"1stFlrSF\",\"2ndFlrSF\",\"TotRmsAbvGrd\"]].copy()\n            tmp[\"Spaciousness\"] = (tmp[\"1stFlrSF\"]+tmp[\"2ndFlrSF\"]) / tmp[\"TotRmsAbvGrd\"]\n            df_out=pd.concat([df_out,tmp[\"Spaciousness\"]],axis=1)\n\n            # 3. External living area (total area of wooden deck, open pouch, fenced pouch, 3-season pouch, screen pouch)\n            tmp = X[[\"WoodDeckSF\",\"OpenPorchSF\",\"EnclosedPorch\",\"3SsnPorch\",\"ScreenPorch\"]].copy()\n            tmp[\"TotalOutsideSF\"] = tmp.sum(axis=1)\n            df_out=pd.concat([df_out,tmp[\"TotalOutsideSF\"]],axis=1)\n\n            # 4. Number of types of outdoor living space\n            tmp = X[[\"WoodDeckSF\",\"OpenPorchSF\",\"EnclosedPorch\",\"3SsnPorch\",\"ScreenPorch\"]].copy()\n            tmp[\"PorchTypes\"] = tmp.gt(0.0).sum(axis=1)\n            df_out=pd.concat([df_out,tmp[\"PorchTypes\"]],axis=1)\n\n            return df_out\n    # 6.Clustering the set of several features\n    clustering_pipe = Pipeline(steps=[(\"impute\", SimpleImputer(strategy=\"median\")),\n                                      ('scaler', StandardScaler()),\n                                      ('KMeans', KMeans(n_clusters=10, n_init=10,random_state=0))])\n    # total pipeline\n    features_pipeline_create = ColumnTransformer(transformers=[\n        (\"numerical\",      numerical_pipe           ,   numerical_cols),\n        (\"categorical_lc\", categorical_lc_pipe      ,   low_cardinality_cols),\n        (\"categorical_hc\", categorical_hc_pipe      ,   high_cardinality_cols),\n        (\"create_features\",CreateFeaturesPipeline() ,   [\"GrLivArea\",\"LotArea\",\n                                                         \"1stFlrSF\",\"2ndFlrSF\",\"TotRmsAbvGrd\",\n                                                         \"WoodDeckSF\",\"OpenPorchSF\",\"EnclosedPorch\",\"3SsnPorch\",\"ScreenPorch\",\n                                                         \"BldgType\"]),\n        (\"clustering_features\", clustering_pipe, [\"LotArea\", \"TotalBsmtSF\", \"1stFlrSF\", \"2ndFlrSF\",\"GrLivArea\"])\n    ])\n    \n\n    # ---Build pipeline dict---------------------------------------------------------------------\n    pipe_dict={\n        f\"{encoder_desc}_01-simple\"           : deepcopy(features_pipeline),\n        f\"{encoder_desc}_02-create-ft\"        : deepcopy(features_pipeline_create),\n    }\n    \n    return pipe_dict\n\n\n# List of tuples. The format of tuple is (encoder_desc,lc_encoder_class,lc_encoder_params,hc_encoder_class,hc_encoder_params.\nencoders=[    \n    (\"01-oe-qe\",   ce.OneHotEncoder,      {\"handle_unknown\":\"value\",\"handle_missing\":\"value\"}, ce.QuantileEncoder,   {\"handle_unknown\":\"value\",\"handle_missing\":\"value\"}),\n    (\"02-oe-te\",   ce.OneHotEncoder,      {\"handle_unknown\":\"value\",\"handle_missing\":\"value\"}, ce.TargetEncoder,     {\"min_samples_leaf\":2,\"smoothing\":1.1,\"handle_unknown\":\"value\",\"handle_missing\":\"value\"}),\n    (\"03-oe-me\",   ce.OneHotEncoder,      {\"handle_unknown\":\"value\",\"handle_missing\":\"value\"}, ce.MEstimateEncoder,  {\"handle_unknown\":\"value\",\"handle_missing\":\"value\"}),\n    # (\"04-oe-se\",   ce.OneHotEncoder,      {\"handle_unknown\":\"value\",\"handle_missing\":\"value\"}, ce.SummaryEncoder,    {\"handle_unknown\":\"value\",\"handle_missing\":\"value\"}),\n    # (\"05-oe-jse\",  ce.OneHotEncoder,      {\"handle_unknown\":\"value\",\"handle_missing\":\"value\"}, ce.JamesSteinEncoder, {\"handle_unknown\":\"value\",\"handle_missing\":\"value\"}),\n]\n\npipe_dict={}\nfor encoder_desc,lc_encoder_class,lc_encoder_params,hc_encoder_class,hc_encoder_params in encoders:\n    tmp_dict = make_pipline_set(encoder_desc,lc_encoder_class,lc_encoder_params,hc_encoder_class,hc_encoder_params)\n    pipe_dict.update(tmp_dict)\n\nprint(\"### builded pipelines \"+\"#\"*30)\nset_config(display=\"diagram\")\nfor pipe_name,pipe in pipe_dict.items():\n    print(pipe_name)\n    display(pipe)\n    print()","metadata":{"papermill":{"duration":0.363917,"end_time":"2022-07-28T12:32:34.009678","exception":false,"start_time":"2022-07-28T12:32:33.645761","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-29T09:57:26.676843Z","iopub.execute_input":"2022-07-29T09:57:26.677259Z","iopub.status.idle":"2022-07-29T09:57:27.465656Z","shell.execute_reply.started":"2022-07-29T09:57:26.677222Z","shell.execute_reply":"2022-07-29T09:57:27.463418Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Chapter: Outlier analysis and exclusion outlier data from training","metadata":{"papermill":{"duration":0.062546,"end_time":"2022-07-28T12:32:34.138103","exception":false,"start_time":"2022-07-28T12:32:34.075557","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"## outlier analysis algorithm\nUnsupervised Outlier Detection using the Local Outlier Factor (LOF).","metadata":{"papermill":{"duration":0.070771,"end_time":"2022-07-28T12:32:34.272164","exception":false,"start_time":"2022-07-28T12:32:34.201393","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"## outlier analysis for the preprocessing output of each pipeline.","metadata":{"papermill":{"duration":0.064634,"end_time":"2022-07-28T12:32:34.410075","exception":false,"start_time":"2022-07-28T12:32:34.345441","status":"completed"},"tags":[]}},{"cell_type":"code","source":"\n\ndef get_inlier_and_outlier_labels(X,y,contamination_rate=0.005):\n    y=y.to_numpy().reshape(-1,1)\n    tmp_df = np.concatenate([X,y],axis=1)\n    labels = LocalOutlierFactor(contamination=contamination_rate).fit_predict(tmp_df)\n    inlier_labels  = labels ==  1\n    outlier_labels = labels == -1\n    print(\"Number of all     samples :\",len(X))\n    print(\"Number of inlier  samples :\",inlier_labels.sum())\n    print(\"Number of outlier samples :\",outlier_labels.sum())\n    return inlier_labels,outlier_labels\n","metadata":{"papermill":{"duration":0.08127,"end_time":"2022-07-28T12:32:34.560135","exception":false,"start_time":"2022-07-28T12:32:34.478865","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-29T09:57:27.467489Z","iopub.execute_input":"2022-07-29T09:57:27.467875Z","iopub.status.idle":"2022-07-29T09:57:27.478483Z","shell.execute_reply.started":"2022-07-29T09:57:27.467844Z","shell.execute_reply":"2022-07-29T09:57:27.476497Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Visualize outliear samples\nAs an example, the result of outlier analysis for the preprocessing output of pipeline '01-ohe-qe_01-simple'。","metadata":{"papermill":{"duration":0.063709,"end_time":"2022-07-28T12:32:34.693409","exception":false,"start_time":"2022-07-28T12:32:34.629700","status":"completed"},"tags":[]}},{"cell_type":"code","source":"TARGET_PIPE_NAME=\"01-oe-qe_01-simple\"\nCONTAMINATION_RATE=0.05\n\npipe=pipe_dict[TARGET_PIPE_NAME]\ntmp_out_X=pipe.fit_transform(train_X,train_y)\ntmp_train_inlier_labels,tmp_train_outlier_labels  = get_inlier_and_outlier_labels(tmp_out_X,train_y,contamination_rate=CONTAMINATION_RATE)\n\ntmp_train_X_inlier  = train_X[tmp_train_inlier_labels]\ntmp_train_X_outlier = train_X[tmp_train_outlier_labels]\ntmp_train_y_inlier  = train_y.to_frame()[tmp_train_inlier_labels]\ntmp_train_y_outlier = train_y.to_frame()[tmp_train_outlier_labels]\n\n\n# objective valiable\nbins = np.histogram(train_y,bins=50)[1]\nfig,axes =plt.subplots(2,1,figsize=(20,10),constrained_layout=True)\nsns.histplot(data=tmp_train_y_inlier,  x=\"SalePrice\",label=\"inlier\",  color=\"teal\",ax=axes[0],bins=bins,edgecolor=\"darkslategray\",linewidth=2)\nsns.histplot(data=tmp_train_y_outlier, x=\"SalePrice\",label=\"outlier\", color=\"magenta\",ax=axes[0],bins=bins,alpha=0.5,edgecolor=\"darkmagenta\",linewidth=2)\nsns.stripplot(data=tmp_train_y_inlier, x=\"SalePrice\",label=\"inlier\",  color=\"teal\",ax=axes[1])\nsns.stripplot(data=tmp_train_y_outlier,x=\"SalePrice\",label=\"outlier\", color=\"darkmagenta\",ax=axes[1])\naxes[0].legend()\naxes[1].legend()\naxes[0].set_title(\"SalePrice\",loc=\"center\",x=0.5,y=0.9)\naxes[1].set_title(\"SalePrice\",loc=\"center\",x=0.5,y=0.9)\naxes[0].set_xlabel(None)\naxes[1].set_xlabel(None)\nplt.show()\n\n\n# features\nn_fig_rows = len(train_X.columns)\nfig,axes =plt.subplots(n_fig_rows,3,figsize=(20,n_fig_rows*2),constrained_layout=True,sharex=\"row\",sharey=\"row\")\nfor ft_no,ft_name in enumerate(train_X.columns):\n    if train_X[ft_name].dtype in [\"object\",\"category\"]:\n        x_order=train_X[ft_name].unique()\n        sns.countplot(data=train_X,          x=ft_name,ax=axes[ft_no,0],label=\"all\",     color=\"darkslategrey\",order=x_order)\n        sns.countplot(data=tmp_train_X_inlier, x=ft_name,ax=axes[ft_no,1],label=\"inlier\",  color=\"teal\",         order=x_order)\n        sns.countplot(data=tmp_train_X_outlier,x=ft_name,ax=axes[ft_no,2],label=\"outlier\", color=\"magenta\",      order=x_order)\n    else:\n        bins=np.histogram(train_X[ft_name].dropna(), bins=50)[1]\n        sns.histplot(train_X[ft_name],           ax=axes[ft_no,0],label=\"all\",     color=\"darkslategrey\",bins=bins)\n        sns.histplot(tmp_train_X_inlier[ft_name],  ax=axes[ft_no,1],label=\"inlier\",  color=\"teal\",         bins=bins)\n        sns.histplot(tmp_train_X_outlier[ft_name], ax=axes[ft_no,2],label=\"outlier\", color=\"magenta\",      bins=bins)\n    \n    axes[ft_no,0].set_title(f\"{ft_name} (all)\"     ,loc=\"center\",x=0.5,y=0.8)\n    axes[ft_no,1].set_title(f\"{ft_name} (inlier)\"  ,loc=\"center\",x=0.5,y=0.8)\n    axes[ft_no,2].set_title(f\"{ft_name} (outlier)\" ,loc=\"center\",x=0.5,y=0.8)\n    axes[ft_no,0].set_xlabel(None)\n    axes[ft_no,1].set_xlabel(None)\n    axes[ft_no,2].set_xlabel(None)\n        \nplt.show()\n","metadata":{"papermill":{"duration":54.529168,"end_time":"2022-07-28T12:33:29.288266","exception":false,"start_time":"2022-07-28T12:32:34.759098","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-29T09:57:27.480814Z","iopub.execute_input":"2022-07-29T09:57:27.481202Z","iopub.status.idle":"2022-07-29T09:58:27.233855Z","shell.execute_reply.started":"2022-07-29T09:57:27.481169Z","shell.execute_reply":"2022-07-29T09:58:27.232601Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Chapter: Data cleansing","metadata":{"papermill":{"duration":0.078201,"end_time":"2022-07-28T12:33:29.446428","exception":false,"start_time":"2022-07-28T12:33:29.368227","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"Implement cleansing procedures in helper functions and apply them to training, validation, and inference data before inputting into the pipeline.\n\nEach time applied to a new dataset,You may encounter new patterns of corrupted data. It needs ad hoc inspection works and considerations, so it should not be included in the pipeline.Therefore, cleansing processes of corrupted data are implemented as helper functions, separated from the pipeline.\n\nTo deal with newly encountered corrupted data, implement updating logics in helper functions. As a result, you can eliminate the redundancy of repeating same works in the next dataset.\n\nHowever, here, for convenience, training and test dataset are combined and dealt at the same time.","metadata":{"papermill":{"duration":0.077382,"end_time":"2022-07-28T12:33:29.602236","exception":false,"start_time":"2022-07-28T12:33:29.524854","status":"completed"},"tags":[]}},{"cell_type":"code","source":"X_full,train_X,train_y = load_csv(os.path.join(INPUT_PATH if EXTERNAL else '/kaggle/input',\"home-data-for-ml-course/train.csv\"),contains_target_column=True)\ntest_X = load_csv(os.path.join(INPUT_PATH if EXTERNAL else '/kaggle/input',\"home-data-for-ml-course/test.csv\"),contains_target_column=False)\n\ntrain_test_X = pd.concat([train_X,test_X])\n\nprint(\"### shape ##########\")\nprint(\"train_X.shape=\",train_X.shape)\nprint(\"train_y.shape=\",train_y.shape)\nprint(\"test_X.shape=\",test_X.shape)\nprint(\"train_test_X.shape=\",train_test_X.shape)\n","metadata":{"papermill":{"duration":0.152354,"end_time":"2022-07-28T12:33:29.833093","exception":false,"start_time":"2022-07-28T12:33:29.680739","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-29T09:58:27.235882Z","iopub.execute_input":"2022-07-29T09:58:27.236357Z","iopub.status.idle":"2022-07-29T09:58:27.321412Z","shell.execute_reply.started":"2022-07-29T09:58:27.236312Z","shell.execute_reply":"2022-07-29T09:58:27.320144Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Compare data to the category list.","metadata":{"papermill":{"duration":0.076993,"end_time":"2022-07-28T12:33:29.987026","exception":false,"start_time":"2022-07-28T12:33:29.910033","status":"completed"},"tags":[]}},{"cell_type":"code","source":"category_dict={\n    \"MSSubClass\":[20,30,40,45,50,60,70,75,80,85,90,120,150,160,180,190,],\n    \"MSZoning\":[\"A\",\"C\",\"FV\",\"I\",\"RH\",\"RL\",\"RP\",\"RM\",],\n    \"Street\":[\"Grvl\",\"Pave\",],\n    \"Alley\":[\"Grvl\",\"Pave\",\"NA\",],\n    \"LotShape\":[\"Reg\",\"IR1\",\"IR2\",\"IR3\",],\n    \"LandContour\":[\"Lvl\",\"Bnk\",\"HLS\",\"Low\",],\n    \"Utilities\":[\"AllPub\",\"NoSewr\",\"NoSeWa\",\"ELO\",],\n    \"LotConfig\":[\"Inside\",\"Corner\",\"CulDSac\",\"FR2\",\"FR3\",],\n    \"LandSlope\":[\"Gtl\",\"Mod\",\"Sev\",],\n    \"Neighborhood\":[\"Blmngtn\",\"Blueste\",\"BrDale\",\"BrkSide\",\"ClearCr\",\"CollgCr\",\"Crawfor\",\"Edwards\",\"Gilbert\",\"IDOTRR\",\"MeadowV\",\"Mitchel\",\"Names\",\"NoRidge\",\"NPkVill\",\"NridgHt\",\"NWAmes\",\"OldTown\",\"SWISU\",\"Sawyer\",\"SawyerW\",\"Somerst\",\"StoneBr\",\"Timber\",\"Veenker\",],\n    \"Condition1\":[\"Artery\",\"Feedr\",\"Norm\",\"RRNn\",\"RRAn\",\"PosN\",\"PosA\",\"RRNe\",\"RRAe\",],\n    \"Condition2\":[\"Artery\",\"Feedr\",\"Norm\",\"RRNn\",\"RRAn\",\"PosN\",\"PosA\",\"RRNe\",\"RRAe\",],\n    \"BldgType\":[\"1Fam\",\"2FmCon\",\"Duplx\",\"TwnhsE\",\"TwnhsI\",],\n    \"HouseStyle\":[\"1Story\",\"1.5Fin\",\"1.5Unf\",\"2Story\",\"2.5Fin\",\"2.5Unf\",\"SFoyer\",\"SLvl\",],\n    \"OverallQual\":[10,9,8,7,6,5,4,3,2,1,],\n    \"OverallCond\":[10,9,8,7,6,5,4,3,2,1,],\n    \"RoofStyle\":[\"Flat\",\"Gable\",\"Gambrel\",\"Hip\",\"Mansard\",\"Shed\",],\n    \"RoofMatl\":[\"ClyTile\",\"CompShg\",\"Membran\",\"Metal\",\"Roll\",\"Tar&Grv\",\"WdShake\",\"WdShngl\",],\n    \"Exterior1st\":[\"AsbShng\",\"AsphShn\",\"BrkComm\",\"BrkFace\",\"CBlock\",\"CemntBd\",\"HdBoard\",\"ImStucc\",\"MetalSd\",\"Other\",\"Plywood\",\"PreCast\",\"Stone\",\"Stucco\",\"VinylSd\",\"Wd Sdng\",\"WdShing\",],\n    \"Exterior2nd\":[\"AsbShng\",\"AsphShn\",\"BrkComm\",\"BrkFace\",\"CBlock\",\"CemntBd\",\"HdBoard\",\"ImStucc\",\"MetalSd\",\"Other\",\"Plywood\",\"PreCast\",\"Stone\",\"Stucco\",\"VinylSd\",\"Wd Sdng\",\"WdShing\",],\n    \"MasVnrType\":[\"BrkCmn\",\"BrkFace\",\"CBlock\",\"None\",\"Stone\",],\n    \"ExterQual\":[\"Ex\",\"Gd\",\"TA\",\"Fa\",\"Po\",],\n    \"ExterCond\":[\"Ex\",\"Gd\",\"TA\",\"Fa\",\"Po\",],\n    \"Foundation\":[\"BrkTil\",\"CBlock\",\"PConc\",\"Slab\",\"Stone\",\"Wood\",],\n    \"BsmtQual\":[\"Ex\",\"Gd\",\"TA\",\"Fa\",\"Po\",\"NA\",],\n    \"BsmtCond\":[\"Ex\",\"Gd\",\"TA\",\"Fa\",\"Po\",\"NA\",],\n    \"BsmtExposure\":[\"Gd\",\"Av\",\"Mn\",\"No\",\"NA\",],\n    \"BsmtFinType1\":[\"GLQ\",\"ALQ\",\"BLQ\",\"Rec\",\"LwQ\",\"Unf\",\"NA\",],\n    \"BsmtFinType2\":[\"GLQ\",\"ALQ\",\"BLQ\",\"Rec\",\"LwQ\",\"Unf\",\"NA\",],\n    \"Heating\":[\"Floor\",\"GasA\",\"GasW\",\"Grav\",\"OthW\",\"Wall\",],\n    \"HeatingQC\":[\"Ex\",\"Gd\",\"TA\",\"Fa\",\"Po\",],\n    \"CentralAir\":[\"N\",\"Y\",],\n    \"Electrical\":[\"SBrkr\",\"FuseA\",\"FuseF\",\"FuseP\",\"Mix\",],\n    \"KitchenQual\":[\"Ex\",\"Gd\",\"TA\",\"Fa\",\"Po\",],\n    \"Functional\":[\"Typ\",\"Min1\",\"Min2\",\"Mod\",\"Maj1\",\"Maj2\",\"Sev\",\"Sal\",],\n    \"FireplaceQu\":[\"Ex\",\"Gd\",\"TA\",\"Fa\",\"Po\",\"NA\",],\n    \"GarageType\":[\"2Types\",\"Attchd\",\"Basment\",\"BuiltIn\",\"CarPort\",\"Detchd\",\"NA\",],\n    \"GarageFinish\":[\"Fin\",\"RFn\",\"Unf\",\"NA\",],\n    \"GarageQual\":[\"Ex\",\"Gd\",\"TA\",\"Fa\",\"Po\",\"NA\",],\n    \"GarageCond\":[\"Ex\",\"Gd\",\"TA\",\"Fa\",\"Po\",\"NA\",],\n    \"PavedDrive\":[\"Y\",\"P\",\"N\",],\n    \"PoolQC\":[\"Ex\",\"Gd\",\"TA\",\"Fa\",\"NA\",],\n    \"Fence\":[\"GdPrv\",\"MnPrv\",\"GdWo\",\"MnWw\",\"NA\",],\n    \"MiscFeature\":[\"Elev\",\"Gar2\",\"Othr\",\"Shed\",\"TenC\",\"NA\",],\n    \"SaleType\":[\"WD\",\"CWD\",\"VWD\",\"New\",\"COD\",\"Con\",\"ConLw\",\"ConLI\",\"ConLD\",\"Oth\",],\n    \"SaleCondition\":[\"Normal\",\"Abnorml\",\"AdjLand\",\"Alloca\",\"Family\",\"Partial\",],\n    \"MoSold\" :[1,2,3,4,5,6,7,8,9,10,11,12],\n    \"YrSold\" :list(range(1870,2021)),\n    \"YearBuilt\":list(range(1870,2021)),\n    \"YearRemodAdd\":list(range(1870,2021)),\n    \"GarageYrBlt\":list(range(1870,2021)),\n}\n\n\ndef compare_category_value(X,y=None):\n    def inner(series_data,category_list):\n        used_values=list(series_data.value_counts().index)\n        diff_list = list(set(used_values)-set(category_list))\n        diff_value_counts = series_data.value_counts()[diff_list]\n        return diff_list,diff_value_counts,used_values\n\n    for key,val in category_dict.items():\n        diff_list,diff_value_counts, used_values=inner(X[key],val)\n        if diff_list:\n            print(\"#####################\")\n            print(\"feature name : \",key)\n            print(\"diff : \",diff_list)\n            print(\"diff_value_counts :\")\n            print(diff_value_counts)\n            print(\"defined categories :\")\n            print(val)\n            print()\n","metadata":{"papermill":{"duration":0.112892,"end_time":"2022-07-28T12:33:30.177320","exception":false,"start_time":"2022-07-28T12:33:30.064428","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-29T09:58:27.323234Z","iopub.execute_input":"2022-07-29T09:58:27.324014Z","iopub.status.idle":"2022-07-29T09:58:27.357959Z","shell.execute_reply.started":"2022-07-29T09:58:27.323968Z","shell.execute_reply":"2022-07-29T09:58:27.356488Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Apply to train and test dataset.","metadata":{"papermill":{"duration":0.076616,"end_time":"2022-07-28T12:33:30.330650","exception":false,"start_time":"2022-07-28T12:33:30.254034","status":"completed"},"tags":[]}},{"cell_type":"code","source":"compare_category_value(train_test_X)","metadata":{"papermill":{"duration":0.191293,"end_time":"2022-07-28T12:33:30.598448","exception":false,"start_time":"2022-07-28T12:33:30.407155","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-29T09:58:27.361909Z","iopub.execute_input":"2022-07-29T09:58:27.362617Z","iopub.status.idle":"2022-07-29T09:58:27.486089Z","shell.execute_reply.started":"2022-07-29T09:58:27.362559Z","shell.execute_reply":"2022-07-29T09:58:27.484969Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Additional ad hoc inspection surrounding features, except for simple typo.","metadata":{"papermill":{"duration":0.077514,"end_time":"2022-07-28T12:33:30.753583","exception":false,"start_time":"2022-07-28T12:33:30.676069","status":"completed"},"tags":[]}},{"cell_type":"code","source":"print(\"### Neighborhood 'NAmes' ##############\")\ndisplay(train_test_X[train_test_X[\"Neighborhood\"]==\"NAmes\"][[\"Condition1\"]].value_counts())\ndisplay(train_test_X[train_test_X[\"Neighborhood\"]==\"NAmes\"][[\"Condition2\"]].value_counts())\nprint()\nprint(\"### Neighborhood 'Names' ##############\")\ndisplay(train_test_X[train_test_X[\"Neighborhood\"]==\"Names\"][[\"Condition1\"]].value_counts())\ndisplay(train_test_X[train_test_X[\"Neighborhood\"]==\"Names\"][[\"Condition2\"]].value_counts())\nprint()\nprint(\"### Neighborhood 'NWAmes' ##############\")\ndisplay(train_test_X[train_test_X[\"Neighborhood\"]==\"NWAmes\"][[\"Condition1\"]].value_counts())\ndisplay(train_test_X[train_test_X[\"Neighborhood\"]==\"NWAmes\"][[\"Condition2\"]].value_counts())\nprint()\nprint(\"### BldgType 'Twnhs' ##############\")\ndisplay(train_test_X[train_test_X[\"BldgType\"]==\"Twnhs\"][[\"MSSubClass\"]].value_counts())\ndisplay(train_test_X[train_test_X[\"BldgType\"]==\"Twnhs\"][[\"MSZoning\"]].value_counts())\ndisplay(train_test_X[train_test_X[\"BldgType\"]==\"Twnhs\"][[\"Utilities\"]].value_counts())\nprint()\nprint(\"### BldgType 'TwnhsE' ##############\")\ndisplay(train_test_X[train_test_X[\"BldgType\"]==\"TwnhsE\"][[\"MSSubClass\"]].value_counts())\ndisplay(train_test_X[train_test_X[\"BldgType\"]==\"TwnhsE\"][[\"MSZoning\"]].value_counts())\ndisplay(train_test_X[train_test_X[\"BldgType\"]==\"TwnhsE\"][[\"Utilities\"]].value_counts())\nprint()\nprint(\"### BldgType 'TwnhsI' ##############\")\ndisplay(train_test_X[train_test_X[\"BldgType\"]==\"TwnhsI\"][[\"MSSubClass\"]].value_counts())\ndisplay(train_test_X[train_test_X[\"BldgType\"]==\"TwnhsI\"][[\"MSZoning\"]].value_counts())\ndisplay(train_test_X[train_test_X[\"BldgType\"]==\"TwnhsI\"][[\"Utilities\"]].value_counts())\nprint()\nprint(\"### Exterior2nd 'Wd Shng' ##############\")\ndisplay(train_test_X[train_test_X[\"Exterior2nd\"]=='Wd Shng'][[\"Exterior1st\"]].value_counts())\nprint()\nprint(\"### Exterior2nd 'Wd Sdng' ##############\")\ndisplay(train_test_X[train_test_X[\"Exterior2nd\"]=='Wd Sdng'][[\"Exterior1st\"]].value_counts())\nprint()\nprint(\"### Exterior2nd 'WdShing' ##############\")\ndisplay(train_test_X[train_test_X[\"Exterior2nd\"]=='WdShing'][[\"Exterior1st\"]].value_counts())\nprint()\nprint(\"### GarageYrBlt 2207.0 ##############\")\ndisplay(train_test_X[train_test_X[\"GarageYrBlt\"]==2207.0][[\"YearBuilt\"]].value_counts())\nprint()\n\n","metadata":{"papermill":{"duration":0.246443,"end_time":"2022-07-28T12:33:31.077530","exception":false,"start_time":"2022-07-28T12:33:30.831087","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-29T09:58:27.487526Z","iopub.execute_input":"2022-07-29T09:58:27.487859Z","iopub.status.idle":"2022-07-29T09:58:27.659538Z","shell.execute_reply.started":"2022-07-29T09:58:27.487829Z","shell.execute_reply":"2022-07-29T09:58:27.658194Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### cleansing plan\n| feature name | corrupted value | overwrite | note|\n|--------------|-----------------|-----------|-----|\n|**MSZoning**|'C (all)'|'C'|typo|\n|**Neighborhood**|'NAmes'|'NWAmes'|Because of non 'Names' sample exists.|\n|**BldgType**|'Duplex'|'Duplx'|typo|\n||'2fmCon'|'2FmCon'|typo|\n||'Twnhs'|'TwnhsE'|Because of non 'TwnhsI' sample exists.|\n|**Exterior2nd**|'Wd Shng'|WdShing|Because of Eterior1st is 'WdShing'.|\n||'Brk Cmn'|'BrkComm'|typo|\n||'CmentBd'|'CemntBd'|typo|\n|**GarageYrBlt**|2207.0|2007|Because of YearBuilt is '2006'|\n\n","metadata":{"papermill":{"duration":0.080934,"end_time":"2022-07-28T12:33:31.239163","exception":false,"start_time":"2022-07-28T12:33:31.158229","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"#### Implement cleansing plan to helper function.","metadata":{"papermill":{"duration":0.080924,"end_time":"2022-07-28T12:33:31.401665","exception":false,"start_time":"2022-07-28T12:33:31.320741","status":"completed"},"tags":[]}},{"cell_type":"code","source":"def cleanse_category(X,y=None):\n    X[\"MSZoning\"]     = X[\"MSZoning\"].replace(      {'C (all)':'C'})\n    X[\"Neighborhood\"] = X[\"Neighborhood\"].replace(  {\"NAmes\"  :\"NWAmes\"})\n    X[\"BldgType\"]     = X[\"BldgType\"].replace(      {\"Duplex\" :\"Duplx\"})\n    X[\"BldgType\"]     = X[\"BldgType\"].replace(      {\"2fmCon\" :\"2FmCon\"})\n    X[\"BldgType\"]     = X[\"BldgType\"].replace(      {\"Twnhs\"  :\"TwnhsE\"})\n    X[\"Exterior2nd\"]  = X[\"Exterior2nd\"].replace(   {\"Wd Shng\":\"WdShing\"})\n    X[\"Exterior2nd\"]  = X[\"Exterior2nd\"].replace(   {\"Brk Cmn\":\"BrkComm\"})\n    X[\"Exterior2nd\"]  = X[\"Exterior2nd\"].replace(   {\"CmentBd\":\"CemntBd\"})\n    X[\"GarageYrBlt\"]  = X[\"GarageYrBlt\"].replace(   {2207.0:2007})\n    return X,y\n    \n# # test code\n# tmp = pd.DataFrame({\n#     \"MSZoning\":['C (all)',np.nan,np.nan],\n#     \"Neighborhood\":[\"NAmes\",np.nan,np.nan],\n#     \"BldgType\":[\"Duplex\",\"2fmCon\",\"Twnhs\"],\n#     \"Exterior2nd\":[\"Wd Shng\",\"Brk Cmn\",\"CmentBd\"]})\n# display(tmp)\n# tmp,_ = cleanse(tmp,None)\n# display(tmp)\n","metadata":{"papermill":{"duration":0.094448,"end_time":"2022-07-28T12:33:31.577723","exception":false,"start_time":"2022-07-28T12:33:31.483275","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-29T09:58:27.661489Z","iopub.execute_input":"2022-07-29T09:58:27.661833Z","iopub.status.idle":"2022-07-29T09:58:27.671197Z","shell.execute_reply.started":"2022-07-29T09:58:27.661802Z","shell.execute_reply":"2022-07-29T09:58:27.669903Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Check the consistency of the time order.","metadata":{"papermill":{"duration":0.081476,"end_time":"2022-07-28T12:33:31.739474","exception":false,"start_time":"2022-07-28T12:33:31.657998","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"relationships between GarageYrBlt and YearBuilt, between YrSold and YearBuilt, between YearRemodAdd and YearBuilt.","metadata":{"papermill":{"duration":0.081201,"end_time":"2022-07-28T12:33:31.901849","exception":false,"start_time":"2022-07-28T12:33:31.820648","status":"completed"},"tags":[]}},{"cell_type":"code","source":"\ndef check_year_diff(X,y=None):\n    def plot_year_diff(df,base_ft_name,target_ft_name,hist_ax,scatter_ax):\n        df_tmp=(df[target_ft_name]-df[base_ft_name]).to_frame()\n        df_tmp.columns=[\"YrDiff\"]\n        df_normal=df[df_tmp.YrDiff>=0]\n        df_anomaly=df[df_tmp.YrDiff<0]\n        print(f\"anomaly counts ({base_ft_name} > {target_ft_name}):\",(df_tmp.YrDiff<0).sum())\n        df[base_ft_name].plot.hist(  ax=hist_ax,alpha=0.5,label=base_ft_name)\n        df[target_ft_name].plot.hist(ax=hist_ax,alpha=0.5,label=target_ft_name)\n        hist_ax.legend()\n        df_normal.plot.scatter( ax=scatter_ax,x=base_ft_name,y=target_ft_name,label=f\"{base_ft_name} <= {target_ft_name}\",color=\"darkblue\")\n        df_anomaly.plot.scatter(ax=scatter_ax,x=base_ft_name,y=target_ft_name,label=f\"{base_ft_name} > {target_ft_name}\",color=\"magenta\")\n        scatter_ax.legend()\n\n    fg,axes=plt.subplots(3,2,figsize=(20,10),constrained_layout=True)\n    plot_year_diff(X,\"YearBuilt\",\"GarageYrBlt\",  axes[0,0],axes[0,1])\n    plot_year_diff(X,\"YearBuilt\",\"YrSold\",       axes[1,0],axes[1,1])\n    plot_year_diff(X,\"YearBuilt\",\"YearRemodAdd\", axes[2,0],axes[2,1])\n    plt.show()\n","metadata":{"papermill":{"duration":0.098916,"end_time":"2022-07-28T12:33:32.083411","exception":false,"start_time":"2022-07-28T12:33:31.984495","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-29T09:58:27.673082Z","iopub.execute_input":"2022-07-29T09:58:27.673461Z","iopub.status.idle":"2022-07-29T09:58:27.687879Z","shell.execute_reply.started":"2022-07-29T09:58:27.673427Z","shell.execute_reply":"2022-07-29T09:58:27.686809Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Apply to training dataset.","metadata":{"papermill":{"duration":0.081162,"end_time":"2022-07-28T12:33:32.245886","exception":false,"start_time":"2022-07-28T12:33:32.164724","status":"completed"},"tags":[]}},{"cell_type":"code","source":"check_year_diff(train_test_X)","metadata":{"papermill":{"duration":2.16997,"end_time":"2022-07-28T12:33:34.497599","exception":false,"start_time":"2022-07-28T12:33:32.327629","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-29T09:58:27.689562Z","iopub.execute_input":"2022-07-29T09:58:27.690029Z","iopub.status.idle":"2022-07-29T09:58:29.708437Z","shell.execute_reply.started":"2022-07-29T09:58:27.689996Z","shell.execute_reply":"2022-07-29T09:58:29.707125Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Additional inspection.","metadata":{"papermill":{"duration":0.281563,"end_time":"2022-07-28T12:33:34.875866","exception":false,"start_time":"2022-07-28T12:33:34.594303","status":"completed"},"tags":[]}},{"cell_type":"code","source":"df_bool=train_test_X[\"YearRemodAdd\"]==1950\ndf_normal=train_test_X[~df_bool]\ndf_anomaly=train_test_X[df_bool]\nprint(\"YearRemodAdd 1950:\",len(df_anomaly))\nfig,ax=plt.subplots(1,1,figsize=(10,5))\ndf_normal.plot.scatter(ax=ax,x=\"YearBuilt\",y=\"YearRemodAdd\",color=\"darkblue\",label=\"YearBuilt <= YearRemodAdd\")\ndf_anomaly.plot.scatter(ax=ax,x=\"YearBuilt\",y=\"YearRemodAdd\",color=\"magenta\",label=\"YearRemodAdd = 1950\")\nax.legend()\nplt.show()","metadata":{"papermill":{"duration":0.452492,"end_time":"2022-07-28T12:33:35.414032","exception":false,"start_time":"2022-07-28T12:33:34.961540","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-29T09:58:29.710130Z","iopub.execute_input":"2022-07-29T09:58:29.710638Z","iopub.status.idle":"2022-07-29T09:58:30.097562Z","shell.execute_reply.started":"2022-07-29T09:58:29.710589Z","shell.execute_reply":"2022-07-29T09:58:30.095492Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### note\n* 18 cases in'GarageYrBlt', 1 case in 'YrSold', 1 case in 'YearRemodAdd' are older than 'YearBuilt'. These may be the correct data, but I think those'are typos in general and I decide to fix those.\n* 361 cases were found in 'YearRemodAdd'. A considerable number of houses built before 1950, have extension and renovation being 1950. This is unnatural, so considered as corrupted data.","metadata":{"papermill":{"duration":0.083552,"end_time":"2022-07-28T12:33:35.581838","exception":false,"start_time":"2022-07-28T12:33:35.498286","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"#### cleansing plan\n| feature name | corrupted value | overwrite | \n|--------------|-----------------|-----------|\n|**GarageYrBlt**|>=df['YearBuilt']|df['YearBuilt']|\n|**YrSold**|>=df['YearBuilt']|df['YearBuilt']|\n|**YrRemodAdd**|>=df['YearBuilt']|df['YearBuilt']|\n|**YrRemodAdd**|df['YearBuilt']==1950|df['YearBuilt']|","metadata":{"papermill":{"duration":0.083375,"end_time":"2022-07-28T12:33:35.749293","exception":false,"start_time":"2022-07-28T12:33:35.665918","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"#### Implement cleansing logics to helper functions.","metadata":{"papermill":{"duration":0.085174,"end_time":"2022-07-28T12:33:35.918652","exception":false,"start_time":"2022-07-28T12:33:35.833478","status":"completed"},"tags":[]}},{"cell_type":"code","source":"def cleanse_year(X,y=None):\n    X[\"GarageYrBlt\" ]= X[\"GarageYrBlt\" ].where(X[\"YearBuilt\"] <= X[\"GarageYrBlt\" ], X[\"YearBuilt\"])\n    X[\"YrSold\"      ]= X[\"YrSold\"      ].where(X[\"YearBuilt\"] <= X[\"YrSold\"      ], X[\"YearBuilt\"])\n    X[\"YearRemodAdd\"]= X[\"YearRemodAdd\"].where(X[\"YearBuilt\"] <= X[\"YearRemodAdd\"], X[\"YearBuilt\"])\n    X[\"YearRemodAdd\"]= X[\"YearRemodAdd\"].where(~(X[\"YearRemodAdd\"]==1950),X[\"YearBuilt\"])\n    return X,y\n    \n# # test code\n# tmp = pd.DataFrame({\n#     \"YearBuilt\"   : [1949,1960,1970],\n#     \"GarageYrBlt\" : [1948,1959,1971],\n#     \"YrSold\"      : [1948,1959,1971],\n#     \"YearRemodAdd\": [1950,1959,1971],   \n# })\n# display(tmp)\n# tmp,_ = cleanse_year(tmp,None)\n# display(tmp)\n","metadata":{"papermill":{"duration":0.09782,"end_time":"2022-07-28T12:33:36.100820","exception":false,"start_time":"2022-07-28T12:33:36.003000","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-29T09:58:30.098910Z","iopub.execute_input":"2022-07-29T09:58:30.100108Z","iopub.status.idle":"2022-07-29T09:58:30.108934Z","shell.execute_reply.started":"2022-07-29T09:58:30.100056Z","shell.execute_reply":"2022-07-29T09:58:30.107342Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Integrate cleansing helpers","metadata":{"papermill":{"duration":0.083815,"end_time":"2022-07-28T12:33:36.267647","exception":false,"start_time":"2022-07-28T12:33:36.183832","status":"completed"},"tags":[]}},{"cell_type":"code","source":"def cleanse_all(X,y=None):\n    X,y = cleanse_category(X,y)\n    X,y = cleanse_year(X,y)\n    return X if y is None else (X,y)\n","metadata":{"papermill":{"duration":0.094889,"end_time":"2022-07-28T12:33:36.446844","exception":false,"start_time":"2022-07-28T12:33:36.351955","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-29T09:58:30.111037Z","iopub.execute_input":"2022-07-29T09:58:30.111552Z","iopub.status.idle":"2022-07-29T09:58:30.126246Z","shell.execute_reply.started":"2022-07-29T09:58:30.111496Z","shell.execute_reply":"2022-07-29T09:58:30.125300Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Chapter: Helpers for Building Models","metadata":{"papermill":{"duration":0.085575,"end_time":"2022-07-28T12:33:36.616002","exception":false,"start_time":"2022-07-28T12:33:36.530427","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"## Helpers of building and testing models","metadata":{"papermill":{"duration":0.084541,"end_time":"2022-07-28T12:33:36.785628","exception":false,"start_time":"2022-07-28T12:33:36.701087","status":"completed"},"tags":[]}},{"cell_type":"code","source":"def fix_seed(seed):\n    # random\n    random.seed(seed)\n    # Numpy\n    np.random.seed(seed)\n    # # Pytorch\n    # torch.manual_seed(seed)\n    # torch.cuda.manual_seed_all(seed)\n    # torch.backends.cudnn.deterministic = True\n    # # Tensorflow\n    # tf.random.set_seed(seed)","metadata":{"papermill":{"duration":0.094195,"end_time":"2022-07-28T12:33:36.964632","exception":false,"start_time":"2022-07-28T12:33:36.870437","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-29T09:58:30.127207Z","iopub.execute_input":"2022-07-29T09:58:30.127546Z","iopub.status.idle":"2022-07-29T09:58:30.138936Z","shell.execute_reply.started":"2022-07-29T09:58:30.127508Z","shell.execute_reply":"2022-07-29T09:58:30.137963Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n# Helper of Prediction for test data\ndef predict_test_dataset(estimator,output_file_name,test_X,display_detail=False):\n    preds = estimator.predict(test_X)\n    output = pd.DataFrame({'Id': test_X.index,'SalePrice': preds})\n    output.to_csv(output_file_name, index=False)\n    print(\"submission file       :\",output_file_name,flush=True)\n\n\n# Helper of building estimator that connects pipeline and model\ndef build_estimator(pipe_name,pipe, model_desc, model,dataset_name,train_X,train_y,test_X,fit_eval_func,**kwargs):\n    estimator_desc = f\"{pipe_name}_{dataset_name}_{model_desc}\"\n    estimator = Pipeline(steps=[(\"preprocess\",deepcopy(pipe)),(\"model\",model)])\n    total_score,estimator = fit_eval_func(estimator,estimator_desc,train_X,train_y,**kwargs)\n    predict_test_dataset(estimator,f\"{estimator_desc}_submission.csv\",test_X)\n    return estimator_desc,estimator,total_score\n","metadata":{"papermill":{"duration":0.097241,"end_time":"2022-07-28T12:33:37.145379","exception":false,"start_time":"2022-07-28T12:33:37.048138","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-29T09:58:30.140101Z","iopub.execute_input":"2022-07-29T09:58:30.140765Z","iopub.status.idle":"2022-07-29T09:58:30.152974Z","shell.execute_reply.started":"2022-07-29T09:58:30.140713Z","shell.execute_reply":"2022-07-29T09:58:30.151710Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Helpers of sequence of fit and prediction using CV, GridSearchCV, Optuna.","metadata":{"papermill":{"duration":0.084899,"end_time":"2022-07-28T12:33:37.313895","exception":false,"start_time":"2022-07-28T12:33:37.228996","status":"completed"},"tags":[]}},{"cell_type":"code","source":"    \n# Helper of sequence of cv and fit using cross_val_score\ndef cv_fit(estimator,estimator_name,X,y,**kwargs):\n    random_state = kwargs[\"random_state\"] if \"random_state\" in kwargs else 0\n    fix_seed(0)\n    score = cross_val_score(estimator,X,y,scoring=\"neg_mean_absolute_error\",cv=KFold(n_splits=5,shuffle=True,random_state=random_state),verbose=0,error_score='raise',n_jobs=-1)\n    total_score= -1 * score.mean()\n    print(f\"{estimator_name} cv mean_score    :\",total_score)\n    estimator.fit(X,y)\n    return total_score,estimator\n\n    \n# Helper of sequence of cv and fit using GridSearchCV\ndef gridsearchcv_fit(estimator,estimator_name,X,y,**kwargs):\n    random_state = kwargs[\"random_state\"] if \"random_state\" in kwargs else 0\n    fix_seed(0)\n    grid = GridSearchCV(estimator,kwargs[\"param_grid\"], verbose=0,scoring=\"neg_mean_absolute_error\",n_jobs=-1,cv=KFold(n_splits=5,shuffle=True,random_state=random_state))\n    sys.stdout.flush()\n    grid.fit(X, y)\n    sys.stdout.flush()\n    print(f\"{estimator_name} gridsearchcv best score :\",-1 * grid.best_score_)\n    print(f\"{estimator_name} gridsearchcv best params:\",grid.best_params_)\n    grid.best_estimator_.fit(X,y)\n    return  -1 * grid.best_score_,grid.best_estimator_\n\n# Helper of sequence of cv and fit using OptunaSearchCV\ndef optuna_cv_fit(estimator,estimator_name,X,y,**kwargs):\n    n_trials = kwargs[\"n_trials\"] if \"n_trials\" in kwargs else 3\n    random_state = kwargs[\"random_state\"] if \"random_state\" in kwargs else 0\n    sys.stdout.flush()\n    fix_seed(0)\n    optuna_search = optuna.integration.OptunaSearchCV(estimator, kwargs[\"params_distributions\"], n_trials=n_trials,\n                                                  verbose=0,random_state=random_state,cv=KFold(n_splits=5,shuffle=True,random_state=1),\n                                                  scoring=\"neg_mean_absolute_error\",refit=True)\n    optuna_search.fit(X,y)\n    sys.stdout.flush()\n    best_score  = -1 * optuna_search.study_.best_trial.value\n    best_params = optuna_search.study_.best_trial.params\n    best_estimator = optuna_search.best_estimator_\n    print(f\"{estimator_name} optuna cv best score:\", best_score)\n    print(f\"{estimator_name} optuna cv best params:\", best_params)\n\n    best_estimator.fit(X, y)\n    return  best_score,best_estimator\n","metadata":{"papermill":{"duration":0.104643,"end_time":"2022-07-28T12:33:37.503012","exception":false,"start_time":"2022-07-28T12:33:37.398369","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-29T09:58:30.154855Z","iopub.execute_input":"2022-07-29T09:58:30.156180Z","iopub.status.idle":"2022-07-29T09:58:30.172625Z","shell.execute_reply.started":"2022-07-29T09:58:30.156140Z","shell.execute_reply":"2022-07-29T09:58:30.171431Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Chapter: Execute all patterns (pipeline and model combinations)","metadata":{"papermill":{"duration":0.086474,"end_time":"2022-07-28T12:33:37.674661","exception":false,"start_time":"2022-07-28T12:33:37.588187","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"## Preparing pipeline dataset pairs( Pipeline + dataset Cutted off outlier data )","metadata":{"papermill":{"duration":0.084949,"end_time":"2022-07-28T12:33:37.845809","exception":false,"start_time":"2022-07-28T12:33:37.760860","status":"completed"},"tags":[]}},{"cell_type":"code","source":"# contamination rate of outlier detecter.\nCONTAMINATION_RATE=0.05\n\n# Reload datasets, to eliminate the possibility of unintentionally updating data during data analysis.\ntrain_X_full,train_X,train_y = load_csv(\"../input/home-data-for-ml-course/train.csv\",contains_target_column=True)\ntest_X = load_csv(\"../input/home-data-for-ml-course/test.csv\",contains_target_column=False)\n\n# Applay cleansing\ntrain_X,train_y = cleanse_all(train_X,train_y)\ntest_X = cleanse_all(test_X)\n\n\n# Reduce sample number for operation test.\n#train_X = train_X[:100]\n#train_y = train_y[:100]\n\n\n\nprint(\"train_X.shape=\",train_X.shape)\nprint(\"train_y.shape=\",train_y.shape)\nprint(\"test_X.shape=\",test_X.shape)\n\npipe_dataset_dict_list = []\n\nfor pipe_name,pipe in pipe_dict.items():\n    for dataset_name in [\"all\",\"inlier\"]:\n        print(f\"### {pipe_name} {dataset_name}###\")\n        if dataset_name==\"inlier\":\n            tmp_out_X=pipe.fit_transform(train_X,train_y)\n            tmp_train_inlier_labels,_  = get_inlier_and_outlier_labels(tmp_out_X,train_y,contamination_rate=CONTAMINATION_RATE)\n            tmp_train_X=train_X[tmp_train_inlier_labels]\n            tmp_train_y=train_y[tmp_train_inlier_labels]\n        else:\n            tmp_train_X=train_X\n            tmp_train_y=train_y\n        \n        pipe_dataset_dict_list.append({\"pipe_name\":pipe_name,\"pipe\":pipe,\"dataset_name\":dataset_name,\"dataset\":(tmp_train_X,tmp_train_y)})\n        print()\n\nprint(\"### summary ###\")\nfor pipe_dataset_dict in pipe_dataset_dict_list:\n    pipe_name    = pipe_dataset_dict[\"pipe_name\"]\n    pipe         = pipe_dataset_dict[\"pipe\"]\n    dataset_name =  pipe_dataset_dict[\"dataset_name\"]\n    tmp_train_X,tmp_train_y = pipe_dataset_dict[\"dataset\"]\n    print(f\"pipe_name={pipe_name}, dataset_name={dataset_name}, train_X.shape={tmp_train_X.shape} ,train_y.shape={tmp_train_y.shape}\")","metadata":{"papermill":{"duration":3.206643,"end_time":"2022-07-28T12:33:41.137454","exception":false,"start_time":"2022-07-28T12:33:37.930811","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-29T09:58:30.173981Z","iopub.execute_input":"2022-07-29T09:58:30.174340Z","iopub.status.idle":"2022-07-29T09:58:35.780424Z","shell.execute_reply.started":"2022-07-29T09:58:30.174307Z","shell.execute_reply":"2022-07-29T09:58:35.778722Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Build estimators and execute training and validation.\neach estimator is built of the combination (pipeline + model).\nAnd training and validation of all estimators are executed with training datasets those make pairs with pipelines.","metadata":{}},{"cell_type":"code","source":"# Reduce trial number of optuna for operation test.\nN_TRIALS=20\nN_RANDOM_STATE_VARIATION=2\n\n\nresult=[]\nestimators={}\nfor pipe_dataset_dict in pipe_dataset_dict_list:\n    pipe_name    = pipe_dataset_dict[\"pipe_name\"]\n    pipe         = pipe_dataset_dict[\"pipe\"]\n    dataset_name =  pipe_dataset_dict[\"dataset_name\"]\n    dataset      = pipe_dataset_dict[\"dataset\"]\n    tmp_train_X,tmp_train_y = dataset\n    print(f\"### {pipe_name}_{dataset_name} \" + (\"#\"*100))\n    \n    # # -----------------------------------------------------------\n    # # RandomForest + cross_val_score\n    # model_name = \"randomforest\"\n    # estimator_name,estimator,score = build_estimator(pipe_name,pipe, model_name, \n    #                                                  RandomForestRegressor(n_estimators=1000,max_depth=10,random_state=1),\n    #                                                  dataset_name,tmp_train_X,tmp_train_y,test_X,cv_fit)\n    # result.append({\"pipe_name\":pipe_name,\"dataset_name\":dataset_name,\"model_name\":model_name,\"estimator_name\":estimator_name,\"score\":score})\n    # estimators[estimator_name]={\"pipe\":pipe,\"dataset\":dataset,\"estimator\":estimator}\n    # print()\n    \n    # # -----------------------------------------------------------\n    # # GradientBoosting + cross_val_score\n    # model_name = \"gradientboosting\"\n    # estimator_name,estimator,score= build_estimator(pipe_name,pipe, model_name,\n    #                                                 GradientBoostingRegressor(n_estimators=1000,max_depth=10,random_state=1),\n    #                                                 dataset_name,tmp_train_X,tmp_train_y,test_X,cv_fit)\n    # result.append({\"pipe_name\":pipe_name,\"dataset_name\":dataset_name,\"model_name\":model_name,\"estimator_name\":estimator_name,\"score\":score})\n    # estimators[estimator_name]={\"pipe\":pipe,\"dataset\":dataset,\"estimator\":estimator}\n    # print()\n\n\n    # -----------------------------------------------------------\n    # XGBoost + OptunaSearchCV, multipled by different random_state\n    for random_state in range(N_RANDOM_STATE_VARIATION):\n        model_name = f\"xgb-optuna-{random_state}\"\n        params_distributions={\n            'model__n_estimators'     : optuna.distributions.IntUniformDistribution(1000,2000),\n            # \"model__learning_rate\"    : optuna.distributions.CategoricalDistribution([0.001,0.005,0.01,0.05,0.1]),\n            # \"model__max_depth\"        : optuna.distributions.IntUniformDistribution(5,10),\n            # \"model__min_child_weight\" : optuna.distributions.IntUniformDistribution(1,5),\n            \"model__gamma\"            : optuna.distributions.UniformDistribution(0.,0.5),\n            \"model__subsample\"        : optuna.distributions.UniformDistribution(0.6,1),\n            \"model__colsample_bytree\" : optuna.distributions.UniformDistribution(0.6,1),\n            \"model__reg_alpha\"        : optuna.distributions.UniformDistribution(1e-5,100),\n            \"model__reg_lambda\"       : optuna.distributions.UniformDistribution(1e-5,1),\n        }\n        params={\"n_trials\":N_TRIALS,\"random_state\":random_state, \"params_distributions\":params_distributions}\n        estimator_name,estimator,score = build_estimator(pipe_name,pipe, model_name,\n                                                         XGBRegressor(random_state=random_state,learning_rate=0.01,max_depth=6,min_child_weight=1),\n                                                         dataset_name,tmp_train_X,tmp_train_y,test_X,optuna_cv_fit,**params)\n        result.append({\"pipe_name\":pipe_name,\"dataset_name\":dataset_name,\"model_name\":model_name,\"estimator_name\":estimator_name,\"score\":score})\n        estimators[estimator_name]={\"pipe\":pipe,\"dataset\":dataset,\"estimator\":estimator}\n        print()\n\ndf_result = pd.DataFrame(result)\ndf_result.index.name=\"exec_order\"","metadata":{"papermill":{"duration":23220.55229,"end_time":"2022-07-28T19:00:41.789554","exception":false,"start_time":"2022-07-28T12:33:41.237264","status":"completed"},"scrolled":true,"tags":[],"execution":{"iopub.status.busy":"2022-07-29T09:58:42.931472Z","iopub.execute_input":"2022-07-29T09:58:42.932062Z","iopub.status.idle":"2022-07-29T10:56:58.776925Z","shell.execute_reply.started":"2022-07-29T09:58:42.932030Z","shell.execute_reply":"2022-07-29T10:56:58.774919Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Visualize results and identify the best score and best estimator (pipeline + train dataset + model).","metadata":{"papermill":{"duration":0.113297,"end_time":"2022-07-28T19:00:42.015680","exception":false,"start_time":"2022-07-28T19:00:41.902383","status":"completed"},"tags":[]}},{"cell_type":"code","source":"N_ENSEMBELE_BASE_ESTIMATORS=10\n\nprint(\"### result summary #############\")\ndisplay(df_result.style.highlight_min(\"score\",color='lightblue',axis=None))\nprint()\n\nprint(\"### best estimaor and submission file #############\")\nbest_estimator      = df_result.loc[df_result[\"score\"].idxmin()][\"estimator_name\"]\nbest_score          = df_result.loc[df_result[\"score\"].idxmin()][\"score\"]\nbest_csv = f\"{best_estimator}_submission.csv\"\nprint(\"best estimator submission csv =\",best_csv)\nprint(\"best estimator score          =\",best_score)\nprint()\n\nprint(f\"### top {N_ENSEMBELE_BASE_ESTIMATORS} estimators #############\")\ndf_top=df_result.sort_values(\"score\",ascending=True)[:N_ENSEMBELE_BASE_ESTIMATORS].copy()\ndf_top=df_top.reset_index(drop=True)\ndf_top.index=df_top.index+1\ndf_top.index.name=\"rank\"\ndisplay(df_top)\ntop_estimators=[estimators[estimator_name]  for estimator_name in df_top[\"estimator_name\"] ]\nprint()\n\nprint(\"### submission files (listing *_submission.csv on working directory.) #############\")\n!ls -la *_submission.csv\n","metadata":{"papermill":{"duration":0.931096,"end_time":"2022-07-28T19:00:43.059686","exception":false,"start_time":"2022-07-28T19:00:42.128590","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-29T10:59:44.721024Z","iopub.execute_input":"2022-07-29T10:59:44.721463Z","iopub.status.idle":"2022-07-29T10:59:45.561607Z","shell.execute_reply.started":"2022-07-29T10:59:44.721427Z","shell.execute_reply":"2022-07-29T10:59:45.560163Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Chapter: Average voting ensemble","metadata":{"papermill":{"duration":0.114015,"end_time":"2022-07-28T19:00:43.287319","exception":false,"start_time":"2022-07-28T19:00:43.173304","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"Build  \"average voting ensemble\"   based on top estimators.<br>\nSince resampling based on outlier judgment is different for each base estimator, I do not take a mechanism to make an ensemble model object (e.g. sklearn.ensemble.VotingRegressor. it should be implemented with fit(X,y) and predict(X), but it can't be implemented with resampling algorithm in scikit-learn's pipeline. Therefore, this method cannot be used. ).\nSo, I build a helper function including the procedure that the sequence (fit and predict) are executed by base models individualy and predictions are averaged finally.","metadata":{"papermill":{"duration":0.112888,"end_time":"2022-07-28T19:00:43.513619","exception":false,"start_time":"2022-07-28T19:00:43.400731","status":"completed"},"tags":[]}},{"cell_type":"code","source":"def average_voting_ensemble(trained_estimators,test_X,weights=None,fit_base_estimators=False):\n    predictions=np.zeros((len(test_X),len(trained_estimators)))\n    for seq,estimator_dict in enumerate(trained_estimators):\n        estimator = estimator_dict[\"estimator\"]\n        if fit_base_estimators:\n            X,y = estimator_dict[\"dataset\"]\n            estimator.fit(X,y)\n        predictions[:,seq] = estimator.predict(test_X)\n    return np.average(predictions,axis=1,weights=weights if weights else [1]*len(trained_estimators))\n\n\ndef create_submission_csv(indices,prediction,filename):\n    output = pd.DataFrame({'Id': indices,'SalePrice': prediction})\n    output.to_csv(filename, index=False)\n    print(\"submission file       :\",filename,flush=True)\n    print()\n    !date\n    !ls -la $filename\n    !head -10 $filename\n\ntest_preds = average_voting_ensemble(top_estimators,test_X)\ncreate_submission_csv(test_X.index,test_preds,\"averaging_ensemble_equally_submission.csv\")\nprint()\n\nweights=list(range(N_ENSEMBELE_BASE_ESTIMATORS,0,-1))\nprint(\"weights=\",weights)\ntest_preds = average_voting_ensemble(top_estimators,test_X,weights)\ncreate_submission_csv(test_X.index,test_preds,\"averaging_ensemble_weighed_submission.csv\")\nprint()\n\nweights=list(range(N_ENSEMBELE_BASE_ESTIMATORS*2,N_ENSEMBELE_BASE_ESTIMATORS,-1))\nprint(\"weights=\",weights)\ntest_preds = average_voting_ensemble(top_estimators,test_X,weights)\ncreate_submission_csv(test_X.index,test_preds,\"averaging_ensemble_gently_weighed_submission.csv\")\n","metadata":{"papermill":{"duration":465.003914,"end_time":"2022-07-28T19:08:28.631191","exception":false,"start_time":"2022-07-28T19:00:43.627277","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-29T11:00:01.686124Z","iopub.execute_input":"2022-07-29T11:00:01.686593Z","iopub.status.idle":"2022-07-29T11:00:14.745758Z","shell.execute_reply.started":"2022-07-29T11:00:01.686557Z","shell.execute_reply":"2022-07-29T11:00:14.744256Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Chapter: Submit","metadata":{"papermill":{"duration":0.114553,"end_time":"2022-07-28T19:08:28.859817","exception":false,"start_time":"2022-07-28T19:08:28.745264","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"Try Submission any or all of the following to get better score.\n* Submission file generated with the single best model.\n* Submission file generated with average voting ensemble(equally averaged).\n* Submission file generated with average voting ensemble(weighted according to ranking).\n* Submission file generated with average voting ensemble(gently weighted according to ranking).\n* Submission file generated with the single model after 2nd place.","metadata":{"papermill":{"duration":0.114374,"end_time":"2022-07-28T19:08:29.090126","exception":false,"start_time":"2022-07-28T19:08:28.975752","status":"completed"},"tags":[]}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}