{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"### kaggle kernels pull eliotbarr/stacking-test-sklearn-xgboost-catboost-lightgbm","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19"}},{"cell_type":"markdown","source":"### Stacking Starter based on Allstate Faron's Script\n\nhttps://www.kaggle.com/mmueller/allstate-claims-severity/stacking-starter/run/390867\n\n### Preprocessing from ogrellier\n\nhttps://www.kaggle.com/ogrellier/good-fun-with-ligthgbm","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nfrom scipy.stats import skew\nimport xgboost as xgb\nfrom sklearn.model_selection import KFold\nfrom sklearn.ensemble import ExtraTreesClassifier\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.metrics import mean_squared_error\nfrom sklearn.linear_model import LogisticRegression\nfrom math import sqrt\nfrom sklearn.metrics import roc_auc_score\nfrom catboost import CatBoostClassifier\nfrom lightgbm import LGBMClassifier\nimport gc\nimport warnings\nwarnings.filterwarnings('ignore')","metadata":{"execution":{"iopub.status.busy":"2022-08-01T00:54:49.919127Z","iopub.execute_input":"2022-08-01T00:54:49.919691Z","iopub.status.idle":"2022-08-01T00:54:53.822694Z","shell.execute_reply.started":"2022-08-01T00:54:49.919574Z","shell.execute_reply":"2022-08-01T00:54:53.820938Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"NFOLDS = 3\nSEED = 0\nNROWS = None","metadata":{"execution":{"iopub.status.busy":"2022-08-01T00:54:53.826154Z","iopub.execute_input":"2022-08-01T00:54:53.826715Z","iopub.status.idle":"2022-08-01T00:54:53.834188Z","shell.execute_reply.started":"2022-08-01T00:54:53.826663Z","shell.execute_reply":"2022-08-01T00:54:53.831992Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data = pd.read_csv('../input/home-credit-default-risk/application_train.csv')\ntest = pd.read_csv('../input/home-credit-default-risk/application_test.csv')\nprev = pd.read_csv('../input/home-credit-default-risk/previous_application.csv')","metadata":{"execution":{"iopub.status.busy":"2022-08-01T00:54:53.836944Z","iopub.execute_input":"2022-08-01T00:54:53.838393Z","iopub.status.idle":"2022-08-01T00:55:22.022277Z","shell.execute_reply.started":"2022-08-01T00:54:53.838299Z","shell.execute_reply":"2022-08-01T00:55:22.020144Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.shape","metadata":{"execution":{"iopub.status.busy":"2022-08-01T00:55:22.025562Z","iopub.execute_input":"2022-08-01T00:55:22.026241Z","iopub.status.idle":"2022-08-01T00:55:22.036204Z","shell.execute_reply.started":"2022-08-01T00:55:22.026200Z","shell.execute_reply":"2022-08-01T00:55:22.033928Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('%-45s | %7s | %10s | %10s | %10s'\n     % ('FEATURES', 'TYPE', 'NB VALUES', 'NB NaNs', 'NaNs (%)'))\n\nfor f_ in data: # .dtpyes\n    print(\"%-45s | %7s | %10s | %10s | %5.2f\"\n          % (f_, str(data[f_].dtype), \n             str(len(data[f_].value_counts(dropna=False))), \n             str(data[f_].isnull().sum()),\n             100 * data[f_].isnull().sum() / data.shape[0])\n         )","metadata":{"execution":{"iopub.status.busy":"2022-08-01T00:55:22.039535Z","iopub.execute_input":"2022-08-01T00:55:22.040778Z","iopub.status.idle":"2022-08-01T00:55:24.539702Z","shell.execute_reply.started":"2022-08-01T00:55:22.040708Z","shell.execute_reply":"2022-08-01T00:55:24.537648Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"categorical_feats = [f for f in data.columns if data[f].dtype == 'object']\n\nfor f_ in categorical_feats:\n    data[f_], indexer = pd.factorize(data[f_])\n    test[f_] = indexer.get_indexer(test[f_])","metadata":{"execution":{"iopub.status.busy":"2022-08-01T00:55:24.542127Z","iopub.execute_input":"2022-08-01T00:55:24.543389Z","iopub.status.idle":"2022-08-01T00:55:25.457778Z","shell.execute_reply.started":"2022-08-01T00:55:24.543334Z","shell.execute_reply":"2022-08-01T00:55:25.456607Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.enable()\n\ny_train = data['TARGET']\ndel data['TARGET']","metadata":{"execution":{"iopub.status.busy":"2022-08-01T00:55:25.459546Z","iopub.execute_input":"2022-08-01T00:55:25.459932Z","iopub.status.idle":"2022-08-01T00:55:25.472006Z","shell.execute_reply.started":"2022-08-01T00:55:25.459896Z","shell.execute_reply":"2022-08-01T00:55:25.470288Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"prev_cat_features = [f_ for f_ in prev.columns if prev[f_].dtype == 'object']\n\nfor f_ in prev_cat_features:\n    prev[f_], _ = pd.factorize(prev[f_])","metadata":{"execution":{"iopub.status.busy":"2022-08-01T00:55:25.473961Z","iopub.execute_input":"2022-08-01T00:55:25.474899Z","iopub.status.idle":"2022-08-01T00:55:30.984378Z","shell.execute_reply.started":"2022-08-01T00:55:25.474848Z","shell.execute_reply":"2022-08-01T00:55:30.983035Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"avg_prev = prev.groupby('SK_ID_CURR').mean()\navg_prev.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-01T00:55:30.986469Z","iopub.execute_input":"2022-08-01T00:55:30.987408Z","iopub.status.idle":"2022-08-01T00:55:33.701210Z","shell.execute_reply.started":"2022-08-01T00:55:30.987357Z","shell.execute_reply":"2022-08-01T00:55:33.699125Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cnt_prev = prev[['SK_ID_CURR', 'SK_ID_PREV']].groupby('SK_ID_CURR').count()\ncnt_prev.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-01T00:55:33.706212Z","iopub.execute_input":"2022-08-01T00:55:33.706667Z","iopub.status.idle":"2022-08-01T00:55:33.981568Z","shell.execute_reply.started":"2022-08-01T00:55:33.706632Z","shell.execute_reply":"2022-08-01T00:55:33.979861Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"avg_prev['nb_app'] = cnt_prev['SK_ID_PREV']","metadata":{"execution":{"iopub.status.busy":"2022-08-01T00:55:33.984343Z","iopub.execute_input":"2022-08-01T00:55:33.984939Z","iopub.status.idle":"2022-08-01T00:55:33.995644Z","shell.execute_reply.started":"2022-08-01T00:55:33.984891Z","shell.execute_reply":"2022-08-01T00:55:33.994098Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del avg_prev['SK_ID_PREV']","metadata":{"execution":{"iopub.status.busy":"2022-08-01T00:55:33.997454Z","iopub.execute_input":"2022-08-01T00:55:33.998708Z","iopub.status.idle":"2022-08-01T00:55:34.005691Z","shell.execute_reply.started":"2022-08-01T00:55:33.998654Z","shell.execute_reply":"2022-08-01T00:55:34.004280Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_train = data.merge(right=avg_prev.reset_index(), how='left', on='SK_ID_CURR')\nx_test = test.merge(right=avg_prev.reset_index(), how='left', on='SK_ID_CURR')\n\nx_train = x_train.fillna(0)\nx_test= x_test.fillna(0)\n\nntrain = x_train.shape[0]\nntest = x_test.shape[0]\n\nexcluded_feats = ['SK_ID_CURR']\nfeatures = [f_ for f_ in x_train.columns if f_ not in excluded_feats]\n\nx_train = x_train[features]\nx_test = x_test[features]","metadata":{"execution":{"iopub.status.busy":"2022-08-01T00:55:34.007184Z","iopub.execute_input":"2022-08-01T00:55:34.008175Z","iopub.status.idle":"2022-08-01T00:55:35.748965Z","shell.execute_reply.started":"2022-08-01T00:55:34.008123Z","shell.execute_reply":"2022-08-01T00:55:35.746947Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_train.shape","metadata":{"execution":{"iopub.status.busy":"2022-08-01T00:55:35.750993Z","iopub.execute_input":"2022-08-01T00:55:35.751548Z","iopub.status.idle":"2022-08-01T00:55:35.759549Z","shell.execute_reply.started":"2022-08-01T00:55:35.751498Z","shell.execute_reply":"2022-08-01T00:55:35.758398Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"kf = KFold(n_splits = NFOLDS, shuffle= True, random_state = SEED)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T00:55:35.761176Z","iopub.execute_input":"2022-08-01T00:55:35.761657Z","iopub.status.idle":"2022-08-01T00:55:35.774876Z","shell.execute_reply.started":"2022-08-01T00:55:35.761619Z","shell.execute_reply":"2022-08-01T00:55:35.773088Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class SklearnWrapper(object):\n    def __init__(self, clf, seed=0, params=None):\n        params['random_state'] = seed\n        self.clf = clf(**params)\n\n    def train(self, x_train, y_train):\n        self.clf.fit(x_train, y_train)\n\n    def predict(self, x):\n        return self.clf.predict_proba(x)[:,1]","metadata":{"execution":{"iopub.status.busy":"2022-08-01T00:55:35.777470Z","iopub.execute_input":"2022-08-01T00:55:35.778048Z","iopub.status.idle":"2022-08-01T00:55:35.789338Z","shell.execute_reply.started":"2022-08-01T00:55:35.778011Z","shell.execute_reply":"2022-08-01T00:55:35.787738Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class CatboostWrapper(object):\n    def __init__(self, clf, seed=0, params=None):\n        params['random_seed'] = seed\n        self.clf = clf(**params)\n\n    def train(self, x_train, y_train):\n        self.clf.fit(x_train, y_train)\n\n    def predict(self, x):\n        return self.clf.predict_proba(x)[:,1]","metadata":{"execution":{"iopub.status.busy":"2022-08-01T00:55:35.791060Z","iopub.execute_input":"2022-08-01T00:55:35.791462Z","iopub.status.idle":"2022-08-01T00:55:35.806789Z","shell.execute_reply.started":"2022-08-01T00:55:35.791431Z","shell.execute_reply":"2022-08-01T00:55:35.805014Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class LightGBMWrapper(object):\n    def __init__(self, clf, seed=0, params=None):\n        params['feature_fraction_seed'] = seed\n        params['bagging_seed'] = seed\n        self.clf = clf(**params)\n\n    def train(self, x_train, y_train):\n        self.clf.fit(x_train, y_train)\n\n    def predict(self, x):\n        return self.clf.predict_proba(x)[:,1]","metadata":{"execution":{"iopub.status.busy":"2022-08-01T00:55:35.809385Z","iopub.execute_input":"2022-08-01T00:55:35.809959Z","iopub.status.idle":"2022-08-01T00:55:35.821630Z","shell.execute_reply.started":"2022-08-01T00:55:35.809900Z","shell.execute_reply":"2022-08-01T00:55:35.819751Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class XgbWrapper(object):\n    def __init__(self, seed=0, params=None):\n        self.param = params\n        self.param['seed'] = seed\n        self.nrounds = params.pop('nrounds', 250)\n\n    def train(self, x_train, y_train):\n        dtrain = xgb.DMatrix(x_train, label=y_train)\n        self.gbdt = xgb.train(self.param, dtrain, self.nrounds)\n\n    def predict(self, x):\n        return self.gbdt.predict(xgb.DMatrix(x))","metadata":{"execution":{"iopub.status.busy":"2022-08-01T00:55:35.824161Z","iopub.execute_input":"2022-08-01T00:55:35.824922Z","iopub.status.idle":"2022-08-01T00:55:35.835721Z","shell.execute_reply.started":"2022-08-01T00:55:35.824695Z","shell.execute_reply":"2022-08-01T00:55:35.834074Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_oof(clf):\n    oof_train = np.zeros((ntrain,))\n    oof_test = np.zeros((ntest,))\n    oof_test_skf = np.empty((NFOLDS, ntest))\n\n    for i, (train_index, test_index) in enumerate(kf.split(x_train)):\n        x_tr = x_train.loc[train_index]\n        y_tr = y_train.loc[train_index]\n        x_te = x_train.loc[test_index]\n\n        clf.train(x_tr, y_tr)\n\n        oof_train[test_index] = clf.predict(x_te)\n        oof_test_skf[i, :] = clf.predict(x_test)\n\n    oof_test[:] = oof_test_skf.mean(axis=0)\n    return oof_train.reshape(-1, 1), oof_test.reshape(-1, 1)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T00:55:35.837749Z","iopub.execute_input":"2022-08-01T00:55:35.838822Z","iopub.status.idle":"2022-08-01T00:55:35.850384Z","shell.execute_reply.started":"2022-08-01T00:55:35.838760Z","shell.execute_reply":"2022-08-01T00:55:35.849093Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"et_params = {\n    'n_jobs': 16,\n    'n_estimators': 200,\n    'max_features': 0.5,\n    'max_depth': 12,\n    'min_samples_leaf': 2,\n}","metadata":{"execution":{"iopub.status.busy":"2022-08-01T00:55:35.852318Z","iopub.execute_input":"2022-08-01T00:55:35.852770Z","iopub.status.idle":"2022-08-01T00:55:35.870422Z","shell.execute_reply.started":"2022-08-01T00:55:35.852734Z","shell.execute_reply":"2022-08-01T00:55:35.868687Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rf_params = {\n    'n_jobs': 16,\n    'n_estimators': 200,\n    'max_features': 0.2,\n    'max_depth': 12,\n    'min_samples_leaf': 2,\n}","metadata":{"execution":{"iopub.status.busy":"2022-08-01T00:55:35.872666Z","iopub.execute_input":"2022-08-01T00:55:35.873178Z","iopub.status.idle":"2022-08-01T00:55:35.884408Z","shell.execute_reply.started":"2022-08-01T00:55:35.873143Z","shell.execute_reply":"2022-08-01T00:55:35.882693Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"xgb_params = {\n    'seed': 0,\n    'colsample_bytree': 0.7,\n    'silent': 1,\n    'subsample': 0.7,\n    'learning_rate': 0.075,\n    'objective': 'binary:logistic',\n    'max_depth': 4,\n    'num_parallel_tree': 1,\n    'min_child_weight': 1,\n    'nrounds': 200\n}","metadata":{"execution":{"iopub.status.busy":"2022-08-01T00:55:35.886181Z","iopub.execute_input":"2022-08-01T00:55:35.886618Z","iopub.status.idle":"2022-08-01T00:55:35.895884Z","shell.execute_reply.started":"2022-08-01T00:55:35.886585Z","shell.execute_reply":"2022-08-01T00:55:35.894585Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"catboost_params = {\n    'iterations': 200,\n    'learning_rate': 0.5,\n    'depth': 3,\n    'l2_leaf_reg': 40,\n    'bootstrap_type': 'Bernoulli',\n    'subsample': 0.7,\n    'scale_pos_weight': 5,\n    'eval_metric': 'AUC',\n    'od_type': 'Iter',\n    'allow_writing_files': False\n}","metadata":{"execution":{"iopub.status.busy":"2022-08-01T00:55:35.897154Z","iopub.execute_input":"2022-08-01T00:55:35.897532Z","iopub.status.idle":"2022-08-01T00:55:35.909899Z","shell.execute_reply.started":"2022-08-01T00:55:35.897501Z","shell.execute_reply":"2022-08-01T00:55:35.908552Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"lightgbm_params = {\n    'n_estimators':200,\n    'learning_rate':0.1,\n    'num_leaves':123,\n    'colsample_bytree':0.8,\n    'subsample':0.9,\n    'max_depth':15,\n    'reg_alpha':0.1,\n    'reg_lambda':0.1,\n    'min_split_gain':0.01,\n    'min_child_weight':2    \n}","metadata":{"execution":{"iopub.status.busy":"2022-08-01T00:55:35.911299Z","iopub.execute_input":"2022-08-01T00:55:35.911806Z","iopub.status.idle":"2022-08-01T00:55:35.923141Z","shell.execute_reply.started":"2022-08-01T00:55:35.911755Z","shell.execute_reply":"2022-08-01T00:55:35.921830Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"xg = XgbWrapper(seed=SEED, params=xgb_params)\net = SklearnWrapper(clf=ExtraTreesClassifier, seed=SEED, params=et_params)\nrf = SklearnWrapper(clf=RandomForestClassifier, seed=SEED, params=rf_params)\ncb = CatboostWrapper(clf= CatBoostClassifier, seed = SEED, params=catboost_params)\nlg = LightGBMWrapper(clf = LGBMClassifier, seed = SEED, params = lightgbm_params)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T00:55:35.924845Z","iopub.execute_input":"2022-08-01T00:55:35.925965Z","iopub.status.idle":"2022-08-01T00:55:35.946462Z","shell.execute_reply.started":"2022-08-01T00:55:35.925902Z","shell.execute_reply":"2022-08-01T00:55:35.945129Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"xg_oof_train, xg_oof_test = get_oof(xg)\net_oof_train, et_oof_test = get_oof(et)\nrf_oof_train, rf_oof_test = get_oof(rf)\ncb_oof_train, cb_oof_test = get_oof(cb)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T00:55:35.950771Z","iopub.execute_input":"2022-08-01T00:55:35.951315Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('XG-CV: {}'.format(sqrt(mean_squared_error(y_train, xg_oof_train))))\nprint('ET-CV: {}'.format(sqrt(mean_squared_error(y_train, et_oof_train))))\nprint('RF-CV: {}'.format(sqrt(mean_squared_error(y_train, rf_oof_train))))\nprint('CB-CV: {}'.format(sqrt(mean_squared_error(y_train, cb_oof_train))))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_train = np.concatenate((xg_oof_train, et_oof_train, rf_oof_train, cb_oof_train), axis=1)\nx_test = np.concatenate((xg_oof_test, et_oof_test, rf_oof_test, cb_oof_test), axis=1)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('{},{}'.format(x_train.shape, x_test.shape))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"logistic_regression = LogisticRegression()\nlogistic_regression.fit(x_train, y_train)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test['TARGET'] = logistic_regression.predict_proba(x_test)[:,1]\n\ntest[['SK_ID_CURR', 'TARGET']].to_csv('first_submission.csv', index=False, float_format='%.8f')","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}