{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport scipy.stats as st\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.metrics import mean_squared_error\n\nimport warnings\nwarnings.filterwarnings('ignore')\n\npd.set_option('display.max_columns', None)\npd.set_option('display.max_rows', 90)","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-31T15:13:04.741497Z","iopub.execute_input":"2022-07-31T15:13:04.741962Z","iopub.status.idle":"2022-07-31T15:13:06.025915Z","shell.execute_reply.started":"2022-07-31T15:13:04.741864Z","shell.execute_reply":"2022-07-31T15:13:06.024801Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n        \n\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"execution":{"iopub.status.busy":"2022-07-31T15:13:06.027986Z","iopub.execute_input":"2022-07-31T15:13:06.028421Z","iopub.status.idle":"2022-07-31T15:13:06.037412Z","shell.execute_reply.started":"2022-07-31T15:13:06.028374Z","shell.execute_reply":"2022-07-31T15:13:06.036537Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pip install feature_engine","metadata":{"execution":{"iopub.status.busy":"2022-07-31T15:13:06.038756Z","iopub.execute_input":"2022-07-31T15:13:06.039315Z","iopub.status.idle":"2022-07-31T15:13:20.493796Z","shell.execute_reply.started":"2022-07-31T15:13:06.039282Z","shell.execute_reply":"2022-07-31T15:13:20.492130Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = pd.read_csv('/kaggle/input/house-prices-advanced-regression-techniques/train.csv')\ndf_test = pd.read_csv('/kaggle/input/house-prices-advanced-regression-techniques/test.csv')\ndf_all = pd.concat([df_train, df_test], axis=0, ignore_index=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-31T15:13:20.497218Z","iopub.execute_input":"2022-07-31T15:13:20.497608Z","iopub.status.idle":"2022-07-31T15:13:20.598593Z","shell.execute_reply.started":"2022-07-31T15:13:20.497572Z","shell.execute_reply":"2022-07-31T15:13:20.597333Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Data size","metadata":{}},{"cell_type":"code","source":"display(df_train.shape)\ndisplay(df_test.shape)\ndisplay(df_all.shape)","metadata":{"execution":{"iopub.status.busy":"2022-07-31T15:13:20.600277Z","iopub.execute_input":"2022-07-31T15:13:20.600614Z","iopub.status.idle":"2022-07-31T15:13:20.614880Z","shell.execute_reply.started":"2022-07-31T15:13:20.600583Z","shell.execute_reply":"2022-07-31T15:13:20.613514Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### I'll concat the train and test to fillna with more quality\n### Saving the target and tests id in variables to not lost them after remove from the all data","metadata":{}},{"cell_type":"code","source":"target = df_train['SalePrice']\ntest_id = df_test['Id']\n\ndf_train = df_all.iloc[0:1460, :]\ndf_test = df_all.iloc[1460:, :]\ndisplay(df_train.shape)\ndisplay(df_test.shape)","metadata":{"execution":{"iopub.status.busy":"2022-07-31T15:13:20.616539Z","iopub.execute_input":"2022-07-31T15:13:20.617331Z","iopub.status.idle":"2022-07-31T15:13:20.641299Z","shell.execute_reply.started":"2022-07-31T15:13:20.617293Z","shell.execute_reply":"2022-07-31T15:13:20.639893Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Looking at the missing data","metadata":{}},{"cell_type":"code","source":"missing = (df_all.isnull().sum() / df_all.shape[0] * 100).sort_values(ascending=False)\nmissing = missing[missing > 0]\nmissing","metadata":{"execution":{"iopub.status.busy":"2022-07-31T15:13:20.643071Z","iopub.execute_input":"2022-07-31T15:13:20.643718Z","iopub.status.idle":"2022-07-31T15:13:20.666931Z","shell.execute_reply.started":"2022-07-31T15:13:20.643686Z","shell.execute_reply":"2022-07-31T15:13:20.665595Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Features with more than 90% missing data will be removed\n##### Alley will not be deleted because their missing values are a category, represents simply \"No alley access\"","metadata":{}},{"cell_type":"code","source":"df_all1 = df_all.drop(['PoolQC', 'MiscFeature', 'Fence', 'SalePrice', 'Id'], axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-07-31T15:13:20.668778Z","iopub.execute_input":"2022-07-31T15:13:20.669526Z","iopub.status.idle":"2022-07-31T15:13:20.683250Z","shell.execute_reply.started":"2022-07-31T15:13:20.669477Z","shell.execute_reply":"2022-07-31T15:13:20.682138Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"missing = (df_all1.isnull().sum() / df_all1.shape[0] * 100).sort_values(ascending=False)\nmissing = missing[missing > 0]\nmissing","metadata":{"execution":{"iopub.status.busy":"2022-07-31T15:13:20.685098Z","iopub.execute_input":"2022-07-31T15:13:20.685649Z","iopub.status.idle":"2022-07-31T15:13:20.706336Z","shell.execute_reply.started":"2022-07-31T15:13:20.685604Z","shell.execute_reply":"2022-07-31T15:13:20.704683Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Ensure Proper Data Types\n#### Possibly there are variables that are numerical but represents caterical values.","metadata":{}},{"cell_type":"code","source":"df_all1.select_dtypes(np.number)","metadata":{"execution":{"iopub.status.busy":"2022-07-31T15:13:20.711496Z","iopub.execute_input":"2022-07-31T15:13:20.711886Z","iopub.status.idle":"2022-07-31T15:13:20.767656Z","shell.execute_reply.started":"2022-07-31T15:13:20.711851Z","shell.execute_reply":"2022-07-31T15:13:20.766554Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### MSSubClass doesn't show us numerical importances\n#### MSSubClass = Identifies the TYPE of dwelling involved in the sale.\n##### It will be transformed to object","metadata":{}},{"cell_type":"code","source":"df_all1['MSSubClass'] = df_all1['MSSubClass'].astype(str)","metadata":{"execution":{"iopub.status.busy":"2022-07-31T15:13:20.769675Z","iopub.execute_input":"2022-07-31T15:13:20.770173Z","iopub.status.idle":"2022-07-31T15:13:20.779578Z","shell.execute_reply.started":"2022-07-31T15:13:20.770127Z","shell.execute_reply":"2022-07-31T15:13:20.778625Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_all1.select_dtypes(np.number)","metadata":{"execution":{"iopub.status.busy":"2022-07-31T15:13:20.780818Z","iopub.execute_input":"2022-07-31T15:13:20.781250Z","iopub.status.idle":"2022-07-31T15:13:20.835066Z","shell.execute_reply.started":"2022-07-31T15:13:20.781205Z","shell.execute_reply":"2022-07-31T15:13:20.833788Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_all1.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-31T15:13:20.836674Z","iopub.execute_input":"2022-07-31T15:13:20.837028Z","iopub.status.idle":"2022-07-31T15:13:20.904166Z","shell.execute_reply.started":"2022-07-31T15:13:20.836997Z","shell.execute_reply":"2022-07-31T15:13:20.902997Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"quantitative = []\nqualitative = []\n\nfor i in df_all1.columns:\n    if df_all1.dtypes[i] != 'object':\n        quantitative.append(i)\n    else:\n        qualitative.append(i)","metadata":{"execution":{"iopub.status.busy":"2022-07-31T15:13:20.905694Z","iopub.execute_input":"2022-07-31T15:13:20.906370Z","iopub.status.idle":"2022-07-31T15:13:20.920790Z","shell.execute_reply.started":"2022-07-31T15:13:20.906322Z","shell.execute_reply":"2022-07-31T15:13:20.919751Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"missing_quali = df_all1[qualitative].isnull().sum().sort_values(ascending=False)\nmissing_quali = missing_quali[missing_quali > 0]\nmissing_quali","metadata":{"execution":{"iopub.status.busy":"2022-07-31T15:13:20.922054Z","iopub.execute_input":"2022-07-31T15:13:20.923009Z","iopub.status.idle":"2022-07-31T15:13:20.943381Z","shell.execute_reply.started":"2022-07-31T15:13:20.922970Z","shell.execute_reply":"2022-07-31T15:13:20.942239Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"missing_quali.index","metadata":{"execution":{"iopub.status.busy":"2022-07-31T15:13:20.944581Z","iopub.execute_input":"2022-07-31T15:13:20.945642Z","iopub.status.idle":"2022-07-31T15:13:20.952387Z","shell.execute_reply.started":"2022-07-31T15:13:20.945602Z","shell.execute_reply":"2022-07-31T15:13:20.951291Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Filling NA values in categorical features\n##### If Na represents some importance, fill it with None.\n##### If just is missing value, I'll fill with mode column value.","metadata":{}},{"cell_type":"code","source":"for i in missing_quali.index:\n    if i in ['Alley',\n                'FireplaceQu',\n                'GarageCond',\n                'GarageQual',\n                'GarageFinish',\n                'GarageType',\n                'BsmtExposure',\n                'BsmtCond',\n                'BsmtQual',\n                'BsmtFinType2',\n                'BsmtFinType1',\n                'MasVnrType']:\n        \n        df_all1[i] = df_all1[i].fillna('None')\n        \n    elif i in ['MSZoning',\n                  'Utilities',\n                  'Functional',\n                  'Exterior2nd',\n                  'KitchenQual',\n                  'Exterior1st',\n                  'Electrical',\n                  'SaleType']:\n      \n        df_all1[i] = df_all1[i].fillna(df_all1[i].mode()[0])\n        \n    else:\n        pass","metadata":{"execution":{"iopub.status.busy":"2022-07-31T15:13:20.953830Z","iopub.execute_input":"2022-07-31T15:13:20.954318Z","iopub.status.idle":"2022-07-31T15:13:20.978830Z","shell.execute_reply.started":"2022-07-31T15:13:20.954286Z","shell.execute_reply":"2022-07-31T15:13:20.977587Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"missing_quali = df_all1[qualitative].isnull().sum().sort_values(ascending=False)\nmissing_quali = missing_quali[missing_quali > 0]\nmissing_quali","metadata":{"execution":{"iopub.status.busy":"2022-07-31T15:13:20.980481Z","iopub.execute_input":"2022-07-31T15:13:20.981555Z","iopub.status.idle":"2022-07-31T15:13:20.999919Z","shell.execute_reply.started":"2022-07-31T15:13:20.981511Z","shell.execute_reply":"2022-07-31T15:13:20.998676Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"missing_quant = (df_all1[quantitative].isnull().sum() / df_all1.shape[0] * 100).sort_values(ascending=False)\nmissing_quant = missing_quant[missing_quant > 0]\nmissing_quant","metadata":{"execution":{"iopub.status.busy":"2022-07-31T15:13:21.001044Z","iopub.execute_input":"2022-07-31T15:13:21.001389Z","iopub.status.idle":"2022-07-31T15:13:21.015857Z","shell.execute_reply.started":"2022-07-31T15:13:21.001358Z","shell.execute_reply":"2022-07-31T15:13:21.014551Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"missing_quant.index","metadata":{"execution":{"iopub.status.busy":"2022-07-31T15:13:21.017718Z","iopub.execute_input":"2022-07-31T15:13:21.018677Z","iopub.status.idle":"2022-07-31T15:13:21.026387Z","shell.execute_reply.started":"2022-07-31T15:13:21.018636Z","shell.execute_reply":"2022-07-31T15:13:21.025315Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in missing_quant.index:\n    if i in ['MasVnrArea',\n             'BsmtHalfBath',\n             'BsmtFullBath',\n             'BsmtUnfSF',\n             'GarageCars',\n             'GarageArea',\n             'TotalBsmtSF',\n             'BsmtFinSF2',\n             'BsmtFinSF1']:\n        \n        df_all1[i] = df_all1[i].fillna(df_all1[i].median())\n        \n    else:\n        pass","metadata":{"execution":{"iopub.status.busy":"2022-07-31T15:13:21.027846Z","iopub.execute_input":"2022-07-31T15:13:21.029149Z","iopub.status.idle":"2022-07-31T15:13:21.047225Z","shell.execute_reply.started":"2022-07-31T15:13:21.029112Z","shell.execute_reply":"2022-07-31T15:13:21.045821Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"missing_quant = (df_all1[quantitative].isnull().sum() / df_all1.shape[0] * 100).sort_values(ascending=False)\nmissing_quant = missing_quant[missing_quant > 0]\nmissing_quant","metadata":{"execution":{"iopub.status.busy":"2022-07-31T15:13:21.049218Z","iopub.execute_input":"2022-07-31T15:13:21.049664Z","iopub.status.idle":"2022-07-31T15:13:21.063499Z","shell.execute_reply.started":"2022-07-31T15:13:21.049622Z","shell.execute_reply":"2022-07-31T15:13:21.062059Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_all2 = df_all1.copy()","metadata":{"execution":{"iopub.status.busy":"2022-07-31T15:13:21.065190Z","iopub.execute_input":"2022-07-31T15:13:21.065952Z","iopub.status.idle":"2022-07-31T15:13:21.070975Z","shell.execute_reply.started":"2022-07-31T15:13:21.065914Z","shell.execute_reply":"2022-07-31T15:13:21.069944Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_all2_number = df_all2.select_dtypes(np.number)\ndf_all2_number.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2022-07-31T15:13:21.072188Z","iopub.execute_input":"2022-07-31T15:13:21.072597Z","iopub.status.idle":"2022-07-31T15:13:21.089146Z","shell.execute_reply.started":"2022-07-31T15:13:21.072568Z","shell.execute_reply":"2022-07-31T15:13:21.087779Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### I'll fill na values using KNN Regressor Imputer\n#### We know KNN works better with normalize data, you can do it now, but I'll wait a little bit","metadata":{}},{"cell_type":"code","source":"from sklearn.impute import KNNImputer\n\nimputer = KNNImputer(n_neighbors=5)\ndf_all2_number = pd.DataFrame(imputer.fit_transform(df_all2_number), columns=df_all2_number.columns)\n\n# Thanks, Kyaw Saw Htoon !!\n# https://medium.com/@kyawsawhtoon/a-guide-to-knn-imputation-95e2dc496e","metadata":{"execution":{"iopub.status.busy":"2022-07-31T15:13:21.090499Z","iopub.execute_input":"2022-07-31T15:13:21.091861Z","iopub.status.idle":"2022-07-31T15:13:21.502742Z","shell.execute_reply.started":"2022-07-31T15:13:21.091810Z","shell.execute_reply":"2022-07-31T15:13:21.501467Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_all2_number.isnull().sum().sum()","metadata":{"execution":{"iopub.status.busy":"2022-07-31T15:13:21.504125Z","iopub.execute_input":"2022-07-31T15:13:21.504499Z","iopub.status.idle":"2022-07-31T15:13:21.514950Z","shell.execute_reply.started":"2022-07-31T15:13:21.504466Z","shell.execute_reply":"2022-07-31T15:13:21.513867Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### How can you see, I hate list comprehensions.\n#### It's not natural haha","metadata":{}},{"cell_type":"code","source":"for i in df_all2_number.columns:\n    df_all2[i] = df_all2_number[i]","metadata":{"execution":{"iopub.status.busy":"2022-07-31T15:13:21.516302Z","iopub.execute_input":"2022-07-31T15:13:21.517420Z","iopub.status.idle":"2022-07-31T15:13:21.531775Z","shell.execute_reply.started":"2022-07-31T15:13:21.517373Z","shell.execute_reply":"2022-07-31T15:13:21.530794Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_all2.isnull().sum().sum()","metadata":{"execution":{"iopub.status.busy":"2022-07-31T15:13:21.537485Z","iopub.execute_input":"2022-07-31T15:13:21.538155Z","iopub.status.idle":"2022-07-31T15:13:21.554937Z","shell.execute_reply.started":"2022-07-31T15:13:21.538114Z","shell.execute_reply":"2022-07-31T15:13:21.554053Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_all3 = df_all2.copy()","metadata":{"execution":{"iopub.status.busy":"2022-07-31T15:13:21.556372Z","iopub.execute_input":"2022-07-31T15:13:21.556985Z","iopub.status.idle":"2022-07-31T15:13:21.563476Z","shell.execute_reply.started":"2022-07-31T15:13:21.556950Z","shell.execute_reply":"2022-07-31T15:13:21.562039Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_all3 = pd.get_dummies(data=df_all3, drop_first=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-31T15:13:21.564971Z","iopub.execute_input":"2022-07-31T15:13:21.565583Z","iopub.status.idle":"2022-07-31T15:13:21.624772Z","shell.execute_reply.started":"2022-07-31T15:13:21.565536Z","shell.execute_reply":"2022-07-31T15:13:21.623542Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_all3.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-31T15:13:21.626202Z","iopub.execute_input":"2022-07-31T15:13:21.626557Z","iopub.status.idle":"2022-07-31T15:13:21.633881Z","shell.execute_reply.started":"2022-07-31T15:13:21.626524Z","shell.execute_reply":"2022-07-31T15:13:21.632677Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Feature Selection\n\n### Codes taught in the 'Feature Selection' course by Professor Soledad Galli (She is amazing !!!)\n#### https://www.udemy.com/share/101XzA3@PHf58nZ1ZlqqmK252r16gjsTY8NlcL9Qzv_uMcVmF5XKGi_qGyaWFHJxqL8l0bWysA==/","metadata":{}},{"cell_type":"code","source":"#pip install feature_engine","metadata":{"execution":{"iopub.status.busy":"2022-07-31T15:13:21.635610Z","iopub.execute_input":"2022-07-31T15:13:21.636073Z","iopub.status.idle":"2022-07-31T15:13:21.644603Z","shell.execute_reply.started":"2022-07-31T15:13:21.636037Z","shell.execute_reply":"2022-07-31T15:13:21.643799Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from feature_engine.selection import DropDuplicateFeatures, DropConstantFeatures, DropCorrelatedFeatures, RecursiveFeatureElimination","metadata":{"execution":{"iopub.status.busy":"2022-07-31T15:13:21.645838Z","iopub.execute_input":"2022-07-31T15:13:21.647201Z","iopub.status.idle":"2022-07-31T15:13:21.704904Z","shell.execute_reply.started":"2022-07-31T15:13:21.647150Z","shell.execute_reply":"2022-07-31T15:13:21.703706Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Searching for Duplicated or Quasi-constant features","metadata":{}},{"cell_type":"code","source":"sel = DropConstantFeatures(tol=0.99, variables=None)\n\nsel.fit(df_all3)\n\nfeatures_to_drop = sel.features_to_drop_\n\nsel.features_to_drop_","metadata":{"execution":{"iopub.status.busy":"2022-07-31T15:13:21.706248Z","iopub.execute_input":"2022-07-31T15:13:21.707441Z","iopub.status.idle":"2022-07-31T15:13:21.898971Z","shell.execute_reply.started":"2022-07-31T15:13:21.707393Z","shell.execute_reply":"2022-07-31T15:13:21.897818Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Duplicated or Quasi-constant will be removed\ndf_all3.drop(features_to_drop, axis=1, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-31T15:13:21.900534Z","iopub.execute_input":"2022-07-31T15:13:21.901238Z","iopub.status.idle":"2022-07-31T15:13:21.908274Z","shell.execute_reply.started":"2022-07-31T15:13:21.901191Z","shell.execute_reply":"2022-07-31T15:13:21.907333Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Investigating Duplicated Features","metadata":{}},{"cell_type":"code","source":"sel = DropDuplicateFeatures(variables=None)\n\nsel.fit(df_all3)\n\nfeatures_to_drop = sel.features_to_drop_\n\nsel.features_to_drop_","metadata":{"execution":{"iopub.status.busy":"2022-07-31T15:13:21.909738Z","iopub.execute_input":"2022-07-31T15:13:21.910306Z","iopub.status.idle":"2022-07-31T15:13:22.606751Z","shell.execute_reply.started":"2022-07-31T15:13:21.910273Z","shell.execute_reply":"2022-07-31T15:13:22.605508Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_all3.drop(features_to_drop, axis=1, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-31T15:13:22.608497Z","iopub.execute_input":"2022-07-31T15:13:22.609237Z","iopub.status.idle":"2022-07-31T15:13:22.616522Z","shell.execute_reply.started":"2022-07-31T15:13:22.609191Z","shell.execute_reply":"2022-07-31T15:13:22.615595Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Looking for Correlated Features","metadata":{}},{"cell_type":"code","source":"sel = DropCorrelatedFeatures(threshold=0.80,    # more than 80% = remove\n                             method='pearson') # you can use other methods like kendall or spearman\n\nsel.fit(df_all3)\n\nfeatures_to_drop = sel.features_to_drop_\n\nprint('Features to drop', sel.features_to_drop_)\nprint()\nprint('Correlated Feature Sets', sel.correlated_feature_sets_)","metadata":{"execution":{"iopub.status.busy":"2022-07-31T15:13:22.617894Z","iopub.execute_input":"2022-07-31T15:13:22.618471Z","iopub.status.idle":"2022-07-31T15:13:23.072811Z","shell.execute_reply.started":"2022-07-31T15:13:22.618437Z","shell.execute_reply":"2022-07-31T15:13:23.071979Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_all3.drop(features_to_drop, axis=1, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-31T15:13:23.074030Z","iopub.execute_input":"2022-07-31T15:13:23.074513Z","iopub.status.idle":"2022-07-31T15:13:23.080101Z","shell.execute_reply.started":"2022-07-31T15:13:23.074481Z","shell.execute_reply":"2022-07-31T15:13:23.079263Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_all3.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-31T15:13:23.081587Z","iopub.execute_input":"2022-07-31T15:13:23.082647Z","iopub.status.idle":"2022-07-31T15:13:23.095980Z","shell.execute_reply.started":"2022-07-31T15:13:23.082214Z","shell.execute_reply":"2022-07-31T15:13:23.094825Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### I'll use GBR to select the features with Recursive Feature Elimination\n### You can run this codes as many times as you find necessary\n### I stop when the r^2 stop to increase\n\n### Codes taught in the 'Feature Selection' course by Professor Soledad Galli (She is amazing !!!)\n#### https://www.udemy.com/share/101XzA3@PHf58nZ1ZlqqmK252r16gjsTY8NlcL9Qzv_uMcVmF5XKGi_qGyaWFHJxqL8l0bWysA==/","metadata":{}},{"cell_type":"code","source":"from sklearn.ensemble import GradientBoostingRegressor\n\n# build initial model using all the features\nmodel = GradientBoostingRegressor(n_estimators=10, max_depth=4, random_state=10)\n\n# Setup the RFE selector\n\nsel = RecursiveFeatureElimination(\n    variables=None, # automatically evaluate all numerical variables\n    estimator = model, # the ML model\n    scoring = 'r2', # the metric we want to evalute\n    threshold = 0.001, # the maximum performance drop allowed to remove a feature\n    cv=5, # cross-validation\n)\n\n# this may take quite a while, because\n# we are building a lot of models with cross-validation\nx = df_all3.iloc[0:1460, :]\n\nsel.fit(x, target)","metadata":{"execution":{"iopub.status.busy":"2022-07-31T15:13:23.098034Z","iopub.execute_input":"2022-07-31T15:13:23.099375Z","iopub.status.idle":"2022-07-31T15:14:35.052508Z","shell.execute_reply.started":"2022-07-31T15:13:23.099323Z","shell.execute_reply":"2022-07-31T15:14:35.051184Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# performance of model trained using all features\n\nsel.initial_model_performance_","metadata":{"execution":{"iopub.status.busy":"2022-07-31T15:14:35.053861Z","iopub.execute_input":"2022-07-31T15:14:35.054201Z","iopub.status.idle":"2022-07-31T15:14:35.060652Z","shell.execute_reply.started":"2022-07-31T15:14:35.054168Z","shell.execute_reply":"2022-07-31T15:14:35.059685Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# importance of all features based of initial model\n\nsel.feature_importances_.plot.bar(figsize=(20,6))\nplt.xlabel('Features')\nplt.ylabel('Importance')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-31T15:14:35.061933Z","iopub.execute_input":"2022-07-31T15:14:35.062399Z","iopub.status.idle":"2022-07-31T15:14:37.328161Z","shell.execute_reply.started":"2022-07-31T15:14:35.062365Z","shell.execute_reply":"2022-07-31T15:14:37.326882Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.Series(sel.performance_drifts_).plot.bar(figsize=(20,6))\nplt.xlabel('Features')\nplt.ylabel('Performance change when feature was added')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-31T15:14:37.329819Z","iopub.execute_input":"2022-07-31T15:14:37.330897Z","iopub.status.idle":"2022-07-31T15:14:39.732282Z","shell.execute_reply.started":"2022-07-31T15:14:37.330855Z","shell.execute_reply":"2022-07-31T15:14:39.730919Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Number of features that will be removed\n\nfeatures_to_drop = sel.features_to_drop_\n\nsel.features_to_drop_","metadata":{"execution":{"iopub.status.busy":"2022-07-31T15:14:39.738671Z","iopub.execute_input":"2022-07-31T15:14:39.739048Z","iopub.status.idle":"2022-07-31T15:14:39.746069Z","shell.execute_reply.started":"2022-07-31T15:14:39.739015Z","shell.execute_reply":"2022-07-31T15:14:39.745003Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_all3.drop(features_to_drop, axis=1, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-31T15:14:39.747435Z","iopub.execute_input":"2022-07-31T15:14:39.748116Z","iopub.status.idle":"2022-07-31T15:14:39.756916Z","shell.execute_reply.started":"2022-07-31T15:14:39.748081Z","shell.execute_reply":"2022-07-31T15:14:39.755786Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.fit(df_all3.iloc[0:1460, :], target)","metadata":{"execution":{"iopub.status.busy":"2022-07-31T15:14:39.758574Z","iopub.execute_input":"2022-07-31T15:14:39.759293Z","iopub.status.idle":"2022-07-31T15:14:39.869592Z","shell.execute_reply.started":"2022-07-31T15:14:39.759255Z","shell.execute_reply":"2022-07-31T15:14:39.868495Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred = model.predict(df_all3.iloc[0:1460, :])","metadata":{"execution":{"iopub.status.busy":"2022-07-31T15:14:39.870947Z","iopub.execute_input":"2022-07-31T15:14:39.871295Z","iopub.status.idle":"2022-07-31T15:14:39.882258Z","shell.execute_reply.started":"2022-07-31T15:14:39.871264Z","shell.execute_reply":"2022-07-31T15:14:39.881044Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import r2_score","metadata":{"execution":{"iopub.status.busy":"2022-07-31T15:14:39.883599Z","iopub.execute_input":"2022-07-31T15:14:39.884590Z","iopub.status.idle":"2022-07-31T15:14:39.889157Z","shell.execute_reply.started":"2022-07-31T15:14:39.884554Z","shell.execute_reply":"2022-07-31T15:14:39.888182Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"r2_final = r2_score(target, y_pred)\nr2_final","metadata":{"execution":{"iopub.status.busy":"2022-07-31T15:14:39.890408Z","iopub.execute_input":"2022-07-31T15:14:39.891261Z","iopub.status.idle":"2022-07-31T15:14:39.903945Z","shell.execute_reply.started":"2022-07-31T15:14:39.891228Z","shell.execute_reply":"2022-07-31T15:14:39.902713Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_submission = pd.read_csv('/kaggle/input/house-prices-advanced-regression-techniques/sample_submission.csv', index_col=False)\nsample_submission","metadata":{"execution":{"iopub.status.busy":"2022-07-31T15:14:39.905511Z","iopub.execute_input":"2022-07-31T15:14:39.905924Z","iopub.status.idle":"2022-07-31T15:14:39.927584Z","shell.execute_reply.started":"2022-07-31T15:14:39.905890Z","shell.execute_reply":"2022-07-31T15:14:39.926427Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = sample_submission.copy()","metadata":{"execution":{"iopub.status.busy":"2022-07-31T15:14:39.929225Z","iopub.execute_input":"2022-07-31T15:14:39.929554Z","iopub.status.idle":"2022-07-31T15:14:39.934050Z","shell.execute_reply.started":"2022-07-31T15:14:39.929522Z","shell.execute_reply":"2022-07-31T15:14:39.933252Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission['SalePrice'] = model.predict(df_all3.iloc[1460:, :])","metadata":{"execution":{"iopub.status.busy":"2022-07-31T15:14:39.935259Z","iopub.execute_input":"2022-07-31T15:14:39.935567Z","iopub.status.idle":"2022-07-31T15:14:39.954684Z","shell.execute_reply.started":"2022-07-31T15:14:39.935537Z","shell.execute_reply":"2022-07-31T15:14:39.953508Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission","metadata":{"execution":{"iopub.status.busy":"2022-07-31T15:14:39.956044Z","iopub.execute_input":"2022-07-31T15:14:39.956567Z","iopub.status.idle":"2022-07-31T15:14:39.970013Z","shell.execute_reply.started":"2022-07-31T15:14:39.956522Z","shell.execute_reply":"2022-07-31T15:14:39.969153Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-31T15:14:39.971620Z","iopub.execute_input":"2022-07-31T15:14:39.972404Z","iopub.status.idle":"2022-07-31T15:14:39.982538Z","shell.execute_reply.started":"2022-07-31T15:14:39.972358Z","shell.execute_reply":"2022-07-31T15:14:39.981789Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_all4 = df_all3.copy()","metadata":{"execution":{"iopub.status.busy":"2022-07-31T15:14:39.983699Z","iopub.execute_input":"2022-07-31T15:14:39.984354Z","iopub.status.idle":"2022-07-31T15:14:39.989283Z","shell.execute_reply.started":"2022-07-31T15:14:39.984319Z","shell.execute_reply":"2022-07-31T15:14:39.988211Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_all4.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-31T15:14:39.990516Z","iopub.execute_input":"2022-07-31T15:14:39.991671Z","iopub.status.idle":"2022-07-31T15:14:40.079032Z","shell.execute_reply.started":"2022-07-31T15:14:39.991628Z","shell.execute_reply":"2022-07-31T15:14:40.077927Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Feature Transformation","metadata":{}},{"cell_type":"code","source":"df_all4.select_dtypes(np.number)","metadata":{"execution":{"iopub.status.busy":"2022-07-31T15:14:40.080638Z","iopub.execute_input":"2022-07-31T15:14:40.081888Z","iopub.status.idle":"2022-07-31T15:14:40.195788Z","shell.execute_reply.started":"2022-07-31T15:14:40.081845Z","shell.execute_reply":"2022-07-31T15:14:40.194744Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Skew Verification","metadata":{}},{"cell_type":"code","source":"skew_df = pd.DataFrame(df_all4.columns, columns=['Feature'])\nskew_df['Skew'] = skew_df['Feature'].apply(lambda feature: st.skew(df_all4[feature]))\nskew_df['Absolute Skew'] = skew_df['Skew'].apply(abs)\nskew_transform = skew_df[skew_df['Absolute Skew'] >= 0.5]\nskew_transform","metadata":{"execution":{"iopub.status.busy":"2022-07-31T15:14:40.197114Z","iopub.execute_input":"2022-07-31T15:14:40.197762Z","iopub.status.idle":"2022-07-31T15:14:40.252299Z","shell.execute_reply.started":"2022-07-31T15:14:40.197694Z","shell.execute_reply":"2022-07-31T15:14:40.251296Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"skew_transform['Feature'].values","metadata":{"execution":{"iopub.status.busy":"2022-07-31T15:14:40.253751Z","iopub.execute_input":"2022-07-31T15:14:40.254136Z","iopub.status.idle":"2022-07-31T15:14:40.261868Z","shell.execute_reply.started":"2022-07-31T15:14:40.254104Z","shell.execute_reply":"2022-07-31T15:14:40.260763Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### I'll use log1p transformation because have a lot of variable with min value = 0\n#### Simple log 0 = infinite =  error","metadata":{}},{"cell_type":"code","source":"for i in skew_transform['Feature'].values:\n    df_all4[i] = np.log1p(df_all4[i])","metadata":{"execution":{"iopub.status.busy":"2022-07-31T15:14:40.263371Z","iopub.execute_input":"2022-07-31T15:14:40.264425Z","iopub.status.idle":"2022-07-31T15:14:40.332436Z","shell.execute_reply.started":"2022-07-31T15:14:40.264382Z","shell.execute_reply":"2022-07-31T15:14:40.331256Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"skew_df = pd.DataFrame(df_all4.columns, columns=['Feature'])\nskew_df['Skew'] = skew_df['Feature'].apply(lambda feature: st.skew(df_all4[feature]))\nskew_df['Absolute Skew'] = skew_df['Skew'].apply(abs)\nskew_df","metadata":{"execution":{"iopub.status.busy":"2022-07-31T15:14:40.334102Z","iopub.execute_input":"2022-07-31T15:14:40.334533Z","iopub.status.idle":"2022-07-31T15:14:40.437812Z","shell.execute_reply.started":"2022-07-31T15:14:40.334489Z","shell.execute_reply":"2022-07-31T15:14:40.436645Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_all4['MoSold'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-07-31T15:14:40.439603Z","iopub.execute_input":"2022-07-31T15:14:40.440669Z","iopub.status.idle":"2022-07-31T15:14:40.449382Z","shell.execute_reply.started":"2022-07-31T15:14:40.440625Z","shell.execute_reply":"2022-07-31T15:14:40.448526Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"-np.cos(0.5236 * 12)","metadata":{"execution":{"iopub.status.busy":"2022-07-31T15:14:40.450903Z","iopub.execute_input":"2022-07-31T15:14:40.451706Z","iopub.status.idle":"2022-07-31T15:14:40.460512Z","shell.execute_reply.started":"2022-07-31T15:14:40.451665Z","shell.execute_reply":"2022-07-31T15:14:40.459788Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_all4['MoSold'] = -np.cos(0.5236 * df_all4['MoSold'])","metadata":{"execution":{"iopub.status.busy":"2022-07-31T15:14:40.462106Z","iopub.execute_input":"2022-07-31T15:14:40.462798Z","iopub.status.idle":"2022-07-31T15:14:40.469970Z","shell.execute_reply.started":"2022-07-31T15:14:40.462754Z","shell.execute_reply":"2022-07-31T15:14:40.469172Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_all4['MoSold'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-07-31T15:14:40.471422Z","iopub.execute_input":"2022-07-31T15:14:40.472202Z","iopub.status.idle":"2022-07-31T15:14:40.482819Z","shell.execute_reply.started":"2022-07-31T15:14:40.472130Z","shell.execute_reply":"2022-07-31T15:14:40.481909Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Scaling","metadata":{}},{"cell_type":"code","source":"df_all5 = df_all4.copy()","metadata":{"execution":{"iopub.status.busy":"2022-07-31T15:14:40.484269Z","iopub.execute_input":"2022-07-31T15:14:40.485252Z","iopub.status.idle":"2022-07-31T15:14:40.493296Z","shell.execute_reply.started":"2022-07-31T15:14:40.485217Z","shell.execute_reply":"2022-07-31T15:14:40.492368Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"scaler = StandardScaler()\ndf_all5 = pd.DataFrame(scaler.fit_transform(df_all5), index=df_all5.index, columns=df_all5.columns)","metadata":{"execution":{"iopub.status.busy":"2022-07-31T15:14:40.495848Z","iopub.execute_input":"2022-07-31T15:14:40.496921Z","iopub.status.idle":"2022-07-31T15:14:40.519259Z","shell.execute_reply.started":"2022-07-31T15:14:40.496872Z","shell.execute_reply":"2022-07-31T15:14:40.517845Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_all5","metadata":{"execution":{"iopub.status.busy":"2022-07-31T15:14:40.520715Z","iopub.execute_input":"2022-07-31T15:14:40.521162Z","iopub.status.idle":"2022-07-31T15:14:40.671355Z","shell.execute_reply.started":"2022-07-31T15:14:40.521119Z","shell.execute_reply":"2022-07-31T15:14:40.670240Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Target Transformation","metadata":{}},{"cell_type":"code","source":"target","metadata":{"execution":{"iopub.status.busy":"2022-07-31T15:14:40.672776Z","iopub.execute_input":"2022-07-31T15:14:40.673090Z","iopub.status.idle":"2022-07-31T15:14:40.680067Z","shell.execute_reply.started":"2022-07-31T15:14:40.673061Z","shell.execute_reply":"2022-07-31T15:14:40.679193Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Here we are notice that the target doesn't have a normal distribuition,\n### I try log distribuition and JohnsonSU distribuition\n### Both are good, but the JohnsonSU is a little better","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(10,6))\nsns.distplot(target, fit=st.norm)","metadata":{"execution":{"iopub.status.busy":"2022-07-31T15:14:40.681237Z","iopub.execute_input":"2022-07-31T15:14:40.681552Z","iopub.status.idle":"2022-07-31T15:14:40.990682Z","shell.execute_reply.started":"2022-07-31T15:14:40.681523Z","shell.execute_reply":"2022-07-31T15:14:40.989630Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(10,6))\nsns.distplot(target, fit=st.johnsonsu)","metadata":{"execution":{"iopub.status.busy":"2022-07-31T15:14:40.992094Z","iopub.execute_input":"2022-07-31T15:14:40.992411Z","iopub.status.idle":"2022-07-31T15:14:41.524082Z","shell.execute_reply.started":"2022-07-31T15:14:40.992382Z","shell.execute_reply":"2022-07-31T15:14:41.522950Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Functions to transform and inverse the johnson's transformations\n# Credits: https://www.kaggle.com/code/lavanyashukla01/how-i-made-top-0-3-on-a-kaggle-competition/notebook\n\ndef johnson(y):\n    gamma, eta, epsilon, lbda = st.johnsonsu.fit(y)\n    target_johnsonsu = gamma + eta*np.arcsinh((y-epsilon)/lbda)\n    return target_johnsonsu, gamma, eta, epsilon, lbda\n            \ndef johnson_inverse(y, gamma, eta, epsilon, lbda):\n    return lbda*np.sinh((y-gamma)/eta) + epsilon","metadata":{"execution":{"iopub.status.busy":"2022-07-31T15:14:41.525625Z","iopub.execute_input":"2022-07-31T15:14:41.526029Z","iopub.status.idle":"2022-07-31T15:14:41.532633Z","shell.execute_reply.started":"2022-07-31T15:14:41.525996Z","shell.execute_reply":"2022-07-31T15:14:41.531508Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"target_johnsonsu, g, et, ep, l = johnson(target)","metadata":{"execution":{"iopub.status.busy":"2022-07-31T15:14:41.534214Z","iopub.execute_input":"2022-07-31T15:14:41.534633Z","iopub.status.idle":"2022-07-31T15:14:41.785109Z","shell.execute_reply.started":"2022-07-31T15:14:41.534593Z","shell.execute_reply":"2022-07-31T15:14:41.783975Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(10,6))\nsns.distplot(target_johnsonsu, fit=st.johnsonsu)","metadata":{"execution":{"iopub.status.busy":"2022-07-31T15:14:41.786502Z","iopub.execute_input":"2022-07-31T15:14:41.787042Z","iopub.status.idle":"2022-07-31T15:14:42.360682Z","shell.execute_reply.started":"2022-07-31T15:14:41.787010Z","shell.execute_reply":"2022-07-31T15:14:42.359587Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"##### Uau, that's good!","metadata":{}},{"cell_type":"code","source":"target_johnsonsu","metadata":{"execution":{"iopub.status.busy":"2022-07-31T15:14:42.361987Z","iopub.execute_input":"2022-07-31T15:14:42.362412Z","iopub.status.idle":"2022-07-31T15:14:42.371811Z","shell.execute_reply.started":"2022-07-31T15:14:42.362380Z","shell.execute_reply":"2022-07-31T15:14:42.370607Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Split Data","metadata":{}},{"cell_type":"code","source":"df_train = df_all5.iloc[0:1460, :]\ndf_test = df_all5.iloc[1460:, :]","metadata":{"execution":{"iopub.status.busy":"2022-07-31T15:25:27.851822Z","iopub.execute_input":"2022-07-31T15:25:27.852673Z","iopub.status.idle":"2022-07-31T15:25:27.858420Z","shell.execute_reply.started":"2022-07-31T15:25:27.852613Z","shell.execute_reply":"2022-07-31T15:25:27.857390Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Model Selection","metadata":{}},{"cell_type":"markdown","source":"### We know that the codes below could easily be done with a simple loop.\n### But for didactic purposes, I will do model by model.","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import cross_val_score, KFold\n\nkfold = KFold(n_splits=10)","metadata":{"execution":{"iopub.status.busy":"2022-07-31T15:25:31.751943Z","iopub.execute_input":"2022-07-31T15:25:31.752365Z","iopub.status.idle":"2022-07-31T15:25:31.758121Z","shell.execute_reply.started":"2022-07-31T15:25:31.752315Z","shell.execute_reply":"2022-07-31T15:25:31.756752Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.svm import SVR\n\nsvr = SVR(kernel='linear')\nsvr.fit(df_train, target_johnsonsu)\ncross_val_svr = cross_val_score(svr, df_train, target_johnsonsu, scoring='neg_mean_squared_error', cv=kfold)\nprint(cross_val_svr, cross_val_svr.std())","metadata":{"execution":{"iopub.status.busy":"2022-07-31T06:22:16.006475Z","iopub.execute_input":"2022-07-31T06:22:16.007541Z","iopub.status.idle":"2022-07-31T06:23:57.718813Z","shell.execute_reply.started":"2022-07-31T06:22:16.007503Z","shell.execute_reply":"2022-07-31T06:23:57.717460Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.linear_model import Lasso\n\nlasso = Lasso()\nlasso.fit(df_train, target_johnsonsu)\ncross_val_lasso = cross_val_score(lasso, df_train, target_johnsonsu, scoring='neg_mean_squared_error', cv=kfold)\nprint(cross_val_lasso.mean(), cross_val_lasso.std())","metadata":{"execution":{"iopub.status.busy":"2022-07-31T06:23:57.720169Z","iopub.execute_input":"2022-07-31T06:23:57.720463Z","iopub.status.idle":"2022-07-31T06:23:57.903257Z","shell.execute_reply.started":"2022-07-31T06:23:57.720432Z","shell.execute_reply":"2022-07-31T06:23:57.901777Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.linear_model import Ridge\n\nridge = Ridge()\nridge.fit(df_train, target_johnsonsu)\ncross_val_ridge = cross_val_score(ridge, df_train, target_johnsonsu, scoring='neg_mean_squared_error', cv=kfold)\nprint(cross_val_ridge.mean(), cross_val_ridge.std())","metadata":{"execution":{"iopub.status.busy":"2022-07-31T06:23:57.905181Z","iopub.execute_input":"2022-07-31T06:23:57.905944Z","iopub.status.idle":"2022-07-31T06:23:58.127443Z","shell.execute_reply.started":"2022-07-31T06:23:57.905882Z","shell.execute_reply":"2022-07-31T06:23:58.125976Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.ensemble import GradientBoostingRegressor\n\ngbr = GradientBoostingRegressor(criterion='mae')\ngbr.fit(df_train, target_johnsonsu)\ncross_val_gbr = cross_val_score(gbr, df_train, target_johnsonsu, scoring='neg_mean_squared_error', cv=kfold)\nprint(cross_val_gbr.mean(), cross_val_gbr.std())","metadata":{"execution":{"iopub.status.busy":"2022-07-31T06:23:58.178880Z","iopub.execute_input":"2022-07-31T06:23:58.183803Z","iopub.status.idle":"2022-07-31T06:31:16.056480Z","shell.execute_reply.started":"2022-07-31T06:23:58.183737Z","shell.execute_reply":"2022-07-31T06:31:16.055433Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.ensemble import RandomForestRegressor\n\nrfr = RandomForestRegressor()\nrfr.fit(df_train, target_johnsonsu)\ncross_val_rfr = cross_val_score(rfr, df_train, target_johnsonsu, scoring='neg_mean_squared_error', cv=kfold)\nprint(cross_val_rfr.mean(), cross_val_rfr.std())","metadata":{"execution":{"iopub.status.busy":"2022-07-31T07:31:48.798188Z","iopub.execute_input":"2022-07-31T07:31:48.798577Z","iopub.status.idle":"2022-07-31T07:32:11.602392Z","shell.execute_reply.started":"2022-07-31T07:31:48.798545Z","shell.execute_reply":"2022-07-31T07:32:11.600981Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.linear_model import BayesianRidge\n\nbayes = BayesianRidge()\nbayes.fit(df_train, target_johnsonsu)\ncross_val_bayes = cross_val_score(bayes, df_train, target_johnsonsu, scoring='neg_mean_squared_error', cv=kfold)\nprint(cross_val_bayes.mean(), cross_val_bayes.std())","metadata":{"execution":{"iopub.status.busy":"2022-07-31T06:31:37.879823Z","iopub.execute_input":"2022-07-31T06:31:37.880133Z","iopub.status.idle":"2022-07-31T06:31:38.786478Z","shell.execute_reply.started":"2022-07-31T06:31:37.880102Z","shell.execute_reply":"2022-07-31T06:31:38.784967Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.neighbors import KNeighborsRegressor\n\nknn = KNeighborsRegressor()\nknn.fit(df_train, target_johnsonsu)\ncross_val_knn = cross_val_score(knn, df_train, target_johnsonsu, scoring='neg_mean_squared_error', cv=kfold)\nprint(cross_val_knn.mean(), cross_val_knn.std())","metadata":{"execution":{"iopub.status.busy":"2022-07-31T06:31:38.788149Z","iopub.execute_input":"2022-07-31T06:31:38.791672Z","iopub.status.idle":"2022-07-31T06:31:39.046758Z","shell.execute_reply.started":"2022-07-31T06:31:38.791615Z","shell.execute_reply":"2022-07-31T06:31:39.045185Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.ensemble import AdaBoostRegressor\n\nadaboost = AdaBoostRegressor(loss='exponential', n_estimators=100)\nadaboost.fit(df_train, target_johnsonsu)\ncross_val_adaboost = cross_val_score(adaboost, df_train, target_johnsonsu, scoring='neg_mean_squared_error', cv=kfold)\nprint(cross_val_adaboost.mean(), cross_val_adaboost.std())","metadata":{"execution":{"iopub.status.busy":"2022-07-31T06:31:39.053680Z","iopub.execute_input":"2022-07-31T06:31:39.057982Z","iopub.status.idle":"2022-07-31T06:31:48.206353Z","shell.execute_reply.started":"2022-07-31T06:31:39.057920Z","shell.execute_reply":"2022-07-31T06:31:48.204970Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.ensemble import ExtraTreesRegressor\n\nextra_tree = ExtraTreesRegressor()\nextra_tree.fit(df_train, target_johnsonsu)\ncross_val_extra_tree = cross_val_score(extra_tree, df_train, target_johnsonsu, scoring='neg_mean_squared_error', cv=kfold)\nprint(cross_val_extra_tree.mean(), cross_val_extra_tree.std())","metadata":{"execution":{"iopub.status.busy":"2022-07-31T07:41:06.490867Z","iopub.execute_input":"2022-07-31T07:41:06.491282Z","iopub.status.idle":"2022-07-31T07:41:31.561875Z","shell.execute_reply.started":"2022-07-31T07:41:06.491248Z","shell.execute_reply":"2022-07-31T07:41:31.560299Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from catboost import CatBoostRegressor\n\ncat_boost = CatBoostRegressor(verbose=0)\ncat_boost.fit(df_train, target_johnsonsu)\ncross_val_cat_boost = cross_val_score(cat_boost, df_train, target_johnsonsu, scoring='neg_mean_squared_error', cv=kfold)\nprint(cross_val_cat_boost.mean(), cross_val_cat_boost.std())","metadata":{"execution":{"iopub.status.busy":"2022-07-31T06:32:11.861900Z","iopub.execute_input":"2022-07-31T06:32:11.862201Z","iopub.status.idle":"2022-07-31T06:32:59.646211Z","shell.execute_reply.started":"2022-07-31T06:32:11.862169Z","shell.execute_reply":"2022-07-31T06:32:59.644796Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.linear_model import OrthogonalMatchingPursuit\n\nomp = OrthogonalMatchingPursuit()\nomp.fit(df_train, target_johnsonsu)\ncross_val_omp = cross_val_score(omp, df_train, target_johnsonsu, scoring='neg_mean_squared_error', cv=kfold)\nprint(cross_val_omp.mean(), cross_val_omp.std())","metadata":{"execution":{"iopub.status.busy":"2022-07-31T06:32:59.647888Z","iopub.execute_input":"2022-07-31T06:32:59.648246Z","iopub.status.idle":"2022-07-31T06:32:59.905211Z","shell.execute_reply.started":"2022-07-31T06:32:59.648212Z","shell.execute_reply":"2022-07-31T06:32:59.901390Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from lightgbm import LGBMRegressor\n\nlightgbm = LGBMRegressor()\nlightgbm.fit(df_train, target_johnsonsu)\ncross_val_lightgbm = cross_val_score(lightgbm, df_train, target_johnsonsu, scoring='neg_mean_squared_error', cv=kfold)\nprint(cross_val_lightgbm.mean(), cross_val_lightgbm.std())","metadata":{"execution":{"iopub.status.busy":"2022-07-31T06:32:59.911980Z","iopub.execute_input":"2022-07-31T06:32:59.916717Z","iopub.status.idle":"2022-07-31T06:33:04.196754Z","shell.execute_reply.started":"2022-07-31T06:32:59.916638Z","shell.execute_reply":"2022-07-31T06:33:04.195378Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Rigde Regression: ', cross_val_ridge.mean(), cross_val_ridge.std())\nprint('Gradient Boost Regressor: ', cross_val_gbr.mean(), cross_val_gbr.std())\nprint('Light Gradient Boost Reg: ', cross_val_lightgbm.mean(), cross_val_lightgbm.std())\nprint('Bayes Ridge: ', cross_val_bayes.mean(), cross_val_bayes.std())\nprint('Cat Boost: ', cross_val_cat_boost.mean(), cross_val_cat_boost.std())","metadata":{"execution":{"iopub.status.busy":"2022-07-31T06:33:04.198470Z","iopub.execute_input":"2022-07-31T06:33:04.199237Z","iopub.status.idle":"2022-07-31T06:33:04.210275Z","shell.execute_reply.started":"2022-07-31T06:33:04.199187Z","shell.execute_reply":"2022-07-31T06:33:04.208595Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.ensemble import GradientBoostingRegressor\nfrom sklearn.model_selection import GridSearchCV\n\nparameters = {'learning_rate': [0.01, 0.02, 0.03, 0.04],\n              'subsample'    : [0.9, 0.5, 0.2, 0.1],\n              'n_estimators' : [1000],\n              'max_depth'    : [2,3,4]}\n\n\ngbr = GradientBoostingRegressor()\n\ngrid_gbr = GridSearchCV(estimator=gbr, param_grid=parameters, cv=5, n_jobs=-1)\ngrid_gbr.fit(df_train, target_johnsonsu)\n\nprint(grid_gbr.best_estimator_)\nprint(grid_gbr.best_score_)","metadata":{"execution":{"iopub.status.busy":"2022-07-31T06:33:04.212754Z","iopub.execute_input":"2022-07-31T06:33:04.213426Z","iopub.status.idle":"2022-07-31T06:33:04.227409Z","shell.execute_reply.started":"2022-07-31T06:33:04.213374Z","shell.execute_reply":"2022-07-31T06:33:04.225799Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'''GradientBoostingRegressor(learning_rate=0.02, n_estimators=1000, subsample=0.5)\n0.912769273397446'''","metadata":{"execution":{"iopub.status.busy":"2022-07-31T06:33:04.229372Z","iopub.execute_input":"2022-07-31T06:33:04.229796Z","iopub.status.idle":"2022-07-31T06:33:04.237692Z","shell.execute_reply.started":"2022-07-31T06:33:04.229748Z","shell.execute_reply":"2022-07-31T06:33:04.236740Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.ensemble import GradientBoostingRegressor\n\ngbr = GradientBoostingRegressor(learning_rate=0.02, subsample=0.5, n_estimators=1000, max_depth=3)\n\ngbr.fit(df_train, target_johnsonsu)","metadata":{"execution":{"iopub.status.busy":"2022-07-31T06:33:04.239308Z","iopub.execute_input":"2022-07-31T06:33:04.239674Z","iopub.status.idle":"2022-07-31T06:33:04.251554Z","shell.execute_reply.started":"2022-07-31T06:33:04.239642Z","shell.execute_reply":"2022-07-31T06:33:04.250315Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred_johnsonsu = svr.predict(df_test)","metadata":{"execution":{"iopub.status.busy":"2022-07-31T06:33:04.253680Z","iopub.execute_input":"2022-07-31T06:33:04.254784Z","iopub.status.idle":"2022-07-31T06:33:04.362045Z","shell.execute_reply.started":"2022-07-31T06:33:04.254715Z","shell.execute_reply":"2022-07-31T06:33:04.361052Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred_johnsonsu","metadata":{"execution":{"iopub.status.busy":"2022-07-31T06:33:04.363161Z","iopub.execute_input":"2022-07-31T06:33:04.364207Z","iopub.status.idle":"2022-07-31T06:33:04.372373Z","shell.execute_reply.started":"2022-07-31T06:33:04.364163Z","shell.execute_reply":"2022-07-31T06:33:04.371146Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred = johnson_inverse(y_pred_johnsonsu, g, et, ep, l)","metadata":{"execution":{"iopub.status.busy":"2022-07-31T06:33:04.373681Z","iopub.execute_input":"2022-07-31T06:33:04.374057Z","iopub.status.idle":"2022-07-31T06:33:04.383743Z","shell.execute_reply.started":"2022-07-31T06:33:04.374026Z","shell.execute_reply":"2022-07-31T06:33:04.382417Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred","metadata":{"execution":{"iopub.status.busy":"2022-07-31T06:33:04.386404Z","iopub.execute_input":"2022-07-31T06:33:04.387515Z","iopub.status.idle":"2022-07-31T06:33:04.398834Z","shell.execute_reply.started":"2022-07-31T06:33:04.387464Z","shell.execute_reply":"2022-07-31T06:33:04.396815Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission2 = sample_submission.copy()","metadata":{"execution":{"iopub.status.busy":"2022-07-31T06:33:04.400430Z","iopub.execute_input":"2022-07-31T06:33:04.400814Z","iopub.status.idle":"2022-07-31T06:33:04.409772Z","shell.execute_reply.started":"2022-07-31T06:33:04.400778Z","shell.execute_reply":"2022-07-31T06:33:04.408740Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission2['SalePrice'] = y_pred","metadata":{"execution":{"iopub.status.busy":"2022-07-31T06:33:04.411481Z","iopub.execute_input":"2022-07-31T06:33:04.413279Z","iopub.status.idle":"2022-07-31T06:33:04.422775Z","shell.execute_reply.started":"2022-07-31T06:33:04.413226Z","shell.execute_reply":"2022-07-31T06:33:04.421357Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission2.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-31T06:33:04.424561Z","iopub.execute_input":"2022-07-31T06:33:04.425138Z","iopub.status.idle":"2022-07-31T06:33:04.444459Z","shell.execute_reply.started":"2022-07-31T06:33:04.425077Z","shell.execute_reply":"2022-07-31T06:33:04.442826Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Submiting the better models and verifing their scores\n\nscore_catboost = ['Catboost', 1-0.12651]\nscore_bayes_ridge = ['Bayes Ridge', 1-0.12964]\nscore_ridge = ['Ridge', 1-0.13204]\nscore_lightgbm = ['Lightgbm', 1-0.13275]\nscore_svr = ['Suport Vector Machine', 1-0.12956]\ntotal_score = score_catboost[1] + score_bayes_ridge[1] + score_ridge[1] + score_lightgbm[1] + score_svr[1]\n                    \n# score_opm = 0.15380 It won't be used.\n# score_gbm = ['Gradient Boost', 0.13575]","metadata":{"execution":{"iopub.status.busy":"2022-07-31T06:33:04.446112Z","iopub.execute_input":"2022-07-31T06:33:04.446849Z","iopub.status.idle":"2022-07-31T06:33:04.454797Z","shell.execute_reply.started":"2022-07-31T06:33:04.446812Z","shell.execute_reply":"2022-07-31T06:33:04.453490Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i, j in [score_catboost, score_bayes_ridge, score_ridge, score_lightgbm, score_svr]:\n    print(f'{i} = {j / total_score * 100:.2f}%')","metadata":{"execution":{"iopub.status.busy":"2022-07-31T06:33:04.456186Z","iopub.execute_input":"2022-07-31T06:33:04.456527Z","iopub.status.idle":"2022-07-31T06:33:04.469421Z","shell.execute_reply.started":"2022-07-31T06:33:04.456495Z","shell.execute_reply":"2022-07-31T06:33:04.468425Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Hyperparameter Optimization","metadata":{}},{"cell_type":"code","source":"'''import optuna'''","metadata":{"execution":{"iopub.status.busy":"2022-07-31T07:33:23.743554Z","iopub.execute_input":"2022-07-31T07:33:23.744473Z","iopub.status.idle":"2022-07-31T07:33:24.354283Z","shell.execute_reply.started":"2022-07-31T07:33:23.744436Z","shell.execute_reply":"2022-07-31T07:33:24.353023Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Define an objective function to be minimized.\n'''def bayes_objective(trial):\n    params = {}\n    n_iter = trial.suggest_int('n_iter', 50, 600)\n    tol = trial.suggest_loguniform('tol', 1e-8, 10.0)\n    alpha_1 = trial.suggest_loguniform('alpha_1', 1e-8, 10.0)\n    alpha_2 = trial.suggest_loguniform('alpha_2', 1e-8, 10.0)\n    lambda_1 = trial.suggest_loguniform('lambda_1', 1e-8, 10.0)\n    lambda_2 = trial.suggest_loguniform('lambda_2', 1e-8, 10.0)\n    \n    model = BayesianRidge(\n        n_iter=n_iter,\n        tol=tol,\n        alpha_1=alpha_1,\n        alpha_2=alpha_2,\n        lambda_1=lambda_1,\n        lambda_2=lambda_2\n    )\n    \n    model.fit(df_train, target_johnsonsu)\n    \n    cv_scores = cross_val_score(model, df_train, target_johnsonsu, scoring='neg_mean_squared_error', cv=kfold)\n    \n    return np.mean(cv_scores)'''","metadata":{"execution":{"iopub.status.busy":"2022-07-31T06:43:43.683397Z","iopub.execute_input":"2022-07-31T06:43:43.683828Z","iopub.status.idle":"2022-07-31T06:43:43.693511Z","shell.execute_reply.started":"2022-07-31T06:43:43.683794Z","shell.execute_reply":"2022-07-31T06:43:43.691924Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'''from sklearn.ensemble import GradientBoostingRegressor\ndef objective(trial):\n\n    ### define params grid to search maximum accuracy\n    \n    loss = trial.suggest_categorical('loss', ['squared_error', 'absolute_error', 'huber'])\n    learning_rate = trial.suggest_loguniform('learning_rate', 1e-2, 1)\n    n_estimators = trial.suggest_int('n_estimators', 50, 300)\n    criterion = trial.suggest_categorical('criterion', ['friedman_mse', 'squared_error'])\n    min_samples_split = trial.suggest_int('min_samples_split', 2, 100)\n    min_samples_leaf = trial.suggest_int('min_samples_leaf', 1, 100)\n    max_depth = trial.suggest_int('max_depth', 1, 100)\n    max_features = trial.suggest_categorical('max_features', ['auto', 'sqrt', 'log2'])\n    \n           \n    ### modeling with suggested params\n    model = GradientBoostingRegressor(loss=loss,\n                                learning_rate = learning_rate,\n                                n_estimators = n_estimators,\n                                criterion = criterion,\n                                min_samples_split =min_samples_split,\n                                min_samples_leaf = min_samples_leaf,\n                                max_depth = max_depth,\n                                max_features = max_features) # do not tune the seed\n\n    ### cross validation score\n    # score = cross_val_score(model, X_train, y_train, n_jobs=-1, cv=3)\n    # etr_score = score.mean()\n\n    ### fit\n    model.fit(df_train, target_johnsonsu) # train on train data\n    cross = cross_val_score(model, df_train, target_johnsonsu, cv=kfold, n_jobs=-1, scoring='neg_mean_squared_error') # validate on validation data\n\n    return np.mean(cross)'''","metadata":{"execution":{"iopub.status.busy":"2022-07-31T08:31:21.071525Z","iopub.execute_input":"2022-07-31T08:31:21.071926Z","iopub.status.idle":"2022-07-31T08:31:21.083540Z","shell.execute_reply.started":"2022-07-31T08:31:21.071888Z","shell.execute_reply":"2022-07-31T08:31:21.082467Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'''study = optuna.create_study(direction='maximize')\nstudy.optimize(objective, n_trials=2000, n_jobs=-1)'''","metadata":{"execution":{"iopub.status.busy":"2022-07-31T14:08:21.135997Z","iopub.execute_input":"2022-07-31T14:08:21.136600Z","iopub.status.idle":"2022-07-31T14:08:21.176305Z","shell.execute_reply.started":"2022-07-31T14:08:21.136469Z","shell.execute_reply":"2022-07-31T14:08:21.174908Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'''study.best_value'''","metadata":{"execution":{"iopub.status.busy":"2022-07-31T12:04:49.699965Z","iopub.execute_input":"2022-07-31T12:04:49.700319Z","iopub.status.idle":"2022-07-31T12:04:49.707088Z","shell.execute_reply.started":"2022-07-31T12:04:49.700289Z","shell.execute_reply":"2022-07-31T12:04:49.706190Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'''study.best_params'''","metadata":{"execution":{"iopub.status.busy":"2022-07-31T12:04:49.708650Z","iopub.execute_input":"2022-07-31T12:04:49.709010Z","iopub.status.idle":"2022-07-31T12:04:49.724518Z","shell.execute_reply.started":"2022-07-31T12:04:49.708979Z","shell.execute_reply":"2022-07-31T12:04:49.723261Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'''# Bayes Ridge\n\n1.34940674596792\n\n{'n_iter': 54,\n 'tol': 0.0003637842501384,\n 'alpha_1': 1.105116339010471e-08,\n 'alpha_2': 9.111163582202963,\n 'lambda_1': 6.8294246189863514,\n 'lambda_2': 1.4041043086396611e-08}'''","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"''''-0.08618141692852478\n\n{'loss': 'huber',\n 'learning_rate': 0.03975240295107851,\n 'n_estimators': 247,\n 'criterion': 'friedman_mse',\n 'min_samples_split': 100,\n 'min_samples_leaf': 17,\n 'max_depth': 97,\n 'max_features': 'sqrt'}''''","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#gbm\n\nparams = {'loss': 'squared_error',\n 'learning_rate': 0.04911616887735104,\n 'n_estimators': 300,\n 'criterion': 'squared_error',\n 'min_samples_split': 92,\n 'min_samples_leaf': 21,\n 'max_depth': 15,\n 'max_features': 'sqrt'}\n\n#-0.08494962289323946","metadata":{"execution":{"iopub.status.busy":"2022-07-31T15:28:09.949283Z","iopub.execute_input":"2022-07-31T15:28:09.950054Z","iopub.status.idle":"2022-07-31T15:28:09.955275Z","shell.execute_reply.started":"2022-07-31T15:28:09.950002Z","shell.execute_reply":"2022-07-31T15:28:09.954535Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.ensemble import GradientBoostingRegressor\n\ngbr = GradientBoostingRegressor(**params)\n\ngbr.fit(df_train, target_johnsonsu)","metadata":{"execution":{"iopub.status.busy":"2022-07-31T15:28:14.575532Z","iopub.execute_input":"2022-07-31T15:28:14.576375Z","iopub.status.idle":"2022-07-31T15:28:15.220284Z","shell.execute_reply.started":"2022-07-31T15:28:14.576337Z","shell.execute_reply":"2022-07-31T15:28:15.219498Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred_johnsonsu = gbr.predict(df_test)","metadata":{"execution":{"iopub.status.busy":"2022-07-31T15:29:23.519308Z","iopub.execute_input":"2022-07-31T15:29:23.519701Z","iopub.status.idle":"2022-07-31T15:29:23.553325Z","shell.execute_reply.started":"2022-07-31T15:29:23.519665Z","shell.execute_reply":"2022-07-31T15:29:23.552199Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred_johnsonsu","metadata":{"execution":{"iopub.status.busy":"2022-07-31T15:31:04.141237Z","iopub.execute_input":"2022-07-31T15:31:04.142225Z","iopub.status.idle":"2022-07-31T15:31:04.150307Z","shell.execute_reply.started":"2022-07-31T15:31:04.142160Z","shell.execute_reply":"2022-07-31T15:31:04.149260Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred = johnson_inverse(y_pred_johnsonsu, g, et, ep, l)","metadata":{"execution":{"iopub.status.busy":"2022-07-31T15:31:07.118029Z","iopub.execute_input":"2022-07-31T15:31:07.118990Z","iopub.status.idle":"2022-07-31T15:31:07.123828Z","shell.execute_reply.started":"2022-07-31T15:31:07.118951Z","shell.execute_reply":"2022-07-31T15:31:07.122584Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred","metadata":{"execution":{"iopub.status.busy":"2022-07-31T15:31:30.020809Z","iopub.execute_input":"2022-07-31T15:31:30.021215Z","iopub.status.idle":"2022-07-31T15:31:30.028255Z","shell.execute_reply.started":"2022-07-31T15:31:30.021180Z","shell.execute_reply":"2022-07-31T15:31:30.027236Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission3 = sample_submission.copy()","metadata":{"execution":{"iopub.status.busy":"2022-07-31T15:31:10.374879Z","iopub.execute_input":"2022-07-31T15:31:10.375270Z","iopub.status.idle":"2022-07-31T15:31:10.380434Z","shell.execute_reply.started":"2022-07-31T15:31:10.375234Z","shell.execute_reply":"2022-07-31T15:31:10.379257Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission3['SalePrice'] = y_pred","metadata":{"execution":{"iopub.status.busy":"2022-07-31T15:31:38.926119Z","iopub.execute_input":"2022-07-31T15:31:38.926887Z","iopub.status.idle":"2022-07-31T15:31:38.933310Z","shell.execute_reply.started":"2022-07-31T15:31:38.926841Z","shell.execute_reply":"2022-07-31T15:31:38.932137Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission3.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-31T15:32:10.360481Z","iopub.execute_input":"2022-07-31T15:32:10.361105Z","iopub.status.idle":"2022-07-31T15:32:10.371314Z","shell.execute_reply.started":"2022-07-31T15:32:10.361070Z","shell.execute_reply":"2022-07-31T15:32:10.370518Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}