{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"Based on [oliver's script](https://www.kaggle.com/ogrellier/xgb-classifier-upsampling-lb-0-283)","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19"}},{"cell_type":"code","source":"MAX_ROUNDS = 400\nOPTIMIZE_ROUNDS = False\nLEARNING_RATE = 0.07\nEARLY_STOPPING_ROUNDS = 50\n# Note : I set EARLY_STOPPING_ROUNDS high so that (when OPTIMIZE_ROUNDS is set)\n#        I will get lots of information to make my own judgment. You should probably\n#        reduce EARLY_STOPPING_ROUNDS if you want to do actual early stopping.","metadata":{"execution":{"iopub.status.busy":"2022-07-11T05:12:42.230041Z","iopub.execute_input":"2022-07-11T05:12:42.230453Z","iopub.status.idle":"2022-07-11T05:12:42.236183Z","shell.execute_reply.started":"2022-07-11T05:12:42.230422Z","shell.execute_reply":"2022-07-11T05:12:42.234891Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"I recommend initially setting MAX_ROUNDS fairly high and using OPTIMIZE_ROUNDS to get an idea of the appropriate number of rounds(which, in my judgment, should be close to the maximum value of best_ntree_limit among all folds, maybe even a bit higher if your model is adequately regularized...or alternatively, you could set verbose = Ture and look at the details to try to find a number of rounds that works well for all folds). Then I would turn off OPTIMIZE_ROUNDS and set MAX_ROUND to the appropriate number of total rounds.\n\nThe problem with \"early stopping\" by choosing the best round for each fold is that it overfits to the validation data. It's therefoere liable not to preduce the optimal model for predicting test data, and if it's used to produce validation data for stacking/ensembling woth ohter models, it would cause this one to have too much weight in the ensemble. Another possibility (and the default for XGBoost, it seems) is to use the round where the early stop actually happens (with the lag that verifies lack of imporvement) rather than the best round. That solves the oerfitting problem (provided the lag is long enough), but so far it doesn't seem to have helped. (I got a worse validation score with 20-round early stopping per fold than with a constant number of rounds for all folds, so the early stopping actually seemed to underfit.)","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nfrom xgboost import XGBClassifier\nfrom sklearn.model_selection import train_test_split, KFold\nfrom sklearn.preprocessing import LabelEncoder\nfrom numba import jit\nimport time\nimport gc\nimport warnings\nwarnings.filterwarnings('ignore')","metadata":{"execution":{"iopub.status.busy":"2022-07-11T05:12:43.194952Z","iopub.execute_input":"2022-07-11T05:12:43.196123Z","iopub.status.idle":"2022-07-11T05:12:43.202012Z","shell.execute_reply.started":"2022-07-11T05:12:43.196089Z","shell.execute_reply":"2022-07-11T05:12:43.200901Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Compute gini\n\n# from CPMP's kernel https://www.kaggle.com/cpmpml/extremely-fast-gini-computation\n\n@jit\ndef eval_gini(y_true, y_prob):\n    y_true = np.asarray(y_true) # numba가 이해할 수 있는 형식으로 변환\n    y_true = y_true[np.argsort(y_prob)]\n    ntrue = 0\n    gini = 0\n    delta = 0\n    n = len(y_true)\n    for i in range(n-1, -1, -1):\n        y_i = y_true[i]\n        ntrue += y_i\n        gini += y_i * delta\n        delta += 1 - y_i\n    gini = 1 - 2 * gini / (ntrue * (n - ntrue))\n    return gini","metadata":{"execution":{"iopub.status.busy":"2022-07-11T05:12:43.717598Z","iopub.execute_input":"2022-07-11T05:12:43.718740Z","iopub.status.idle":"2022-07-11T05:12:43.727340Z","shell.execute_reply.started":"2022-07-11T05:12:43.718660Z","shell.execute_reply":"2022-07-11T05:12:43.726297Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Functions from oliver's kernel\n# https://www.kaggle.com/ogrellier/xgb-classifier-upsampling-lb-0-283\n\ndef gini_xgb(preds, dtrain):\n    labels = dtrain.get_label()\n    gini_score = -eval_gini(labels, preds)\n    return [('gini', gini_score)]\n\n\ndef add_noise(series, noise_level):\n    return series * (1 + noise_level * np.random.randn(len(series)))\n\n\ndef target_encode(trn_series=None,    # Revised to encode validation series\n                  val_series=None,\n                  tst_series=None,\n                  target=None,\n                  min_samples_leaf=1,\n                  smoothing=1,\n                  noise_level=0):\n    \"\"\"\n    Smoothing is computed like in the following paper by Daniele Micci-Barreca\n    https://kaggle2.blob.core.windows.net/forum-message-attachments/225952/7441/high%20cardinality%20categoricals.pdf\n    trn_series : training categorical feature as a pd.Series\n    tst_series : test categorical feature as a pd.Series\n    target : target data as a pd.Series\n    min_samples_leaf (int) : minimum samples to take category average into account\n    smoothing (int) : smoothing effect to balance categorical average vs prior\n    \"\"\"\n    assert len(trn_series) == len(target)\n    assert trn_series.name == tst_series.name\n    temp = pd.concat([trn_series, target], axis=1)\n    # Compute target mean\n    averages = temp.groupby(by=trn_series.name)[target.name].agg([\"mean\", \"count\"])\n    \n    # Compute smoothing\n    smoothing = 1 / (1 + np.exp(-(averages[\"count\"] - min_samples_leaf) / smoothing))\n    \n    # Apply average function to all target data\n    prior = target.mean()\n    \n    # The bigger the count the less full_avg is taken into account\n    averages[target.name] = prior * (1 - smoothing) + averages[\"mean\"] * smoothing\n    averages.drop([\"mean\", \"count\"], axis=1, inplace=True)\n    \n    # Apply averages to trn and tst series\n    ft_trn_series = pd.merge(\n        trn_series.to_frame(trn_series.name),\n        averages.reset_index().rename(columns={'index': target.name, target.name: 'average'}),\n        on=trn_series.name,\n        how='left')['average'].rename(trn_series.name + '_mean').fillna(prior)\n    # pd.merge does not keep the index so restore it\n    ft_trn_series.index = trn_series.index\n    \n    ft_val_series = pd.merge(\n        val_series.to_frame(val_series.name),\n        averages.reset_index().rename(columns={'index': target.name, target.name: 'average'}),\n        on=val_series.name,\n        how='left')['average'].rename(trn_series.name + '_mean').fillna(prior)\n    # pd.merge does not keep the index so restore it\n    ft_val_series.index = val_series.index\n    \n    ft_tst_series = pd.merge(\n        tst_series.to_frame(tst_series.name),\n        averages.reset_index().rename(columns={'index': target.name, target.name: 'average'}),\n        on=tst_series.name,\n        how='left')['average'].rename(trn_series.name + '_mean').fillna(prior)\n    # pd.merge does not keep the index so restore it\n    ft_tst_series.index = tst_series.index\n    \n    return add_noise(ft_trn_series, noise_level), add_noise(ft_val_series, noise_level), add_noise(ft_tst_series, noise_level)","metadata":{"execution":{"iopub.status.busy":"2022-07-11T05:12:44.378694Z","iopub.execute_input":"2022-07-11T05:12:44.379399Z","iopub.status.idle":"2022-07-11T05:12:44.397957Z","shell.execute_reply.started":"2022-07-11T05:12:44.379359Z","shell.execute_reply":"2022-07-11T05:12:44.396627Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Read data\ntrain_df = pd.read_csv('../input/porto-seguro-safe-driver-prediction/train.csv', na_values= '-1')  # iloc[0:200, :]\ntest_df = pd.read_csv('../input/porto-seguro-safe-driver-prediction/test.csv', na_values= '-1')","metadata":{"execution":{"iopub.status.busy":"2022-07-11T05:12:44.743938Z","iopub.execute_input":"2022-07-11T05:12:44.744706Z","iopub.status.idle":"2022-07-11T05:12:52.530784Z","shell.execute_reply.started":"2022-07-11T05:12:44.744667Z","shell.execute_reply":"2022-07-11T05:12:52.529939Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df","metadata":{"execution":{"iopub.status.busy":"2022-07-11T05:12:52.532208Z","iopub.execute_input":"2022-07-11T05:12:52.532927Z","iopub.status.idle":"2022-07-11T05:12:52.678918Z","shell.execute_reply.started":"2022-07-11T05:12:52.532896Z","shell.execute_reply":"2022-07-11T05:12:52.677599Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# from olivier\ntrain_features = [\n    \"ps_car_13\",  #            : 1571.65 / shadow  609.23\n    \"ps_reg_03\",  #            : 1408.42 / shadow  511.15\n    \"ps_ind_05_cat\",  #        : 1387.87 / shadow   84.72\n    \"ps_ind_03\",  #            : 1219.47 / shadow  230.55\n    \"ps_ind_15\",  #            :  922.18 / shadow  242.00\n    \"ps_reg_02\",  #            :  920.65 / shadow  267.50\n    \"ps_car_14\",  #            :  798.48 / shadow  549.58\n    \"ps_car_12\",  #            :  731.93 / shadow  293.62\n    \"ps_car_01_cat\",  #        :  698.07 / shadow  178.72\n    \"ps_car_07_cat\",  #        :  694.53 / shadow   36.35\n    \"ps_ind_17_bin\",  #        :  620.77 / shadow   23.15\n    \"ps_car_03_cat\",  #        :  611.73 / shadow   50.67\n    \"ps_reg_01\",  #            :  598.60 / shadow  178.57\n    \"ps_car_15\",  #            :  593.35 / shadow  226.43\n    \"ps_ind_01\",  #            :  547.32 / shadow  154.58\n    \"ps_ind_16_bin\",  #        :  475.37 / shadow   34.17\n    \"ps_ind_07_bin\",  #        :  435.28 / shadow   28.92\n    \"ps_car_06_cat\",  #        :  398.02 / shadow  212.43\n    \"ps_car_04_cat\",  #        :  376.87 / shadow   76.98\n    \"ps_ind_06_bin\",  #        :  370.97 / shadow   36.13\n    \"ps_car_09_cat\",  #        :  214.12 / shadow   81.38\n    \"ps_car_02_cat\",  #        :  203.03 / shadow   26.67\n    \"ps_ind_02_cat\",  #        :  189.47 / shadow   65.68\n    \"ps_car_11\",  #            :  173.28 / shadow   76.45\n    \"ps_car_05_cat\",  #        :  172.75 / shadow   62.92\n    \"ps_calc_09\",  #           :  169.13 / shadow  129.72\n    \"ps_calc_05\",  #           :  148.83 / shadow  120.68\n    \"ps_ind_08_bin\",  #        :  140.73 / shadow   27.63\n    \"ps_car_08_cat\",  #        :  120.87 / shadow   28.82\n    \"ps_ind_09_bin\",  #        :  113.92 / shadow   27.05\n    \"ps_ind_04_cat\",  #        :  107.27 / shadow   37.43\n    \"ps_ind_18_bin\",  #        :   77.42 / shadow   25.97\n    \"ps_ind_12_bin\",  #        :   39.67 / shadow   15.52\n    \"ps_ind_14\",  #            :   37.37 / shadow   16.65\n]\n\n# add combination\ncombs = [ ('ps_reg_01', 'ps_car_02_cat'),\n         ('ps_reg_01', 'ps_car_04_cat'),]","metadata":{"execution":{"iopub.status.busy":"2022-07-11T05:12:52.681510Z","iopub.execute_input":"2022-07-11T05:12:52.681970Z","iopub.status.idle":"2022-07-11T05:12:52.691613Z","shell.execute_reply.started":"2022-07-11T05:12:52.681924Z","shell.execute_reply":"2022-07-11T05:12:52.690103Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Process data\nid_test = test_df['id'].values\nid_train = train_df['id'].values\ny = train_df['target']\n\nstart = time.time()\nfor n_c, (f1, f2) in enumerate(combs):\n    name1 = f1 + \"_plus_\" + f2\n    print('current feature %60s %4d in %5.1f'\n          % (name1, n_c + 1, (time.time() - start) / 60), end='')\n    print('\\r' * 75, end='')\n    train_df[name1] = train_df[f1].apply(lambda x: str(x)) + \"_\" + train_df[f2].apply(lambda x: str(x))\n    test_df[name1] = test_df[f1].apply(lambda x: str(x)) + \"_\" + test_df[f2].apply(lambda x: str(x))\n    # Label Encode\n    lbl = LabelEncoder()\n    lbl.fit(list(train_df[name1].values) + list(test_df[name1].values))\n    train_df[name1] = lbl.transform(list(train_df[name1].values))\n    test_df[name1] = lbl.transform(list(test_df[name1].values))\n\n    train_features.append(name1)\n    \nX = train_df[train_features]\ntest_df = test_df[train_features]\n\nf_cats = [f for f in X.columns if \"_cat\" in f]","metadata":{"execution":{"iopub.status.busy":"2022-07-11T05:12:52.695101Z","iopub.execute_input":"2022-07-11T05:12:52.695574Z","iopub.status.idle":"2022-07-11T05:13:04.522132Z","shell.execute_reply.started":"2022-07-11T05:12:52.695541Z","shell.execute_reply":"2022-07-11T05:13:04.520459Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_valid_pred = 0*y\ny_test_pred = 0","metadata":{"execution":{"iopub.status.busy":"2022-07-11T05:13:04.523521Z","iopub.execute_input":"2022-07-11T05:13:04.523843Z","iopub.status.idle":"2022-07-11T05:13:04.530142Z","shell.execute_reply.started":"2022-07-11T05:13:04.523815Z","shell.execute_reply":"2022-07-11T05:13:04.529060Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Set up folds\nK = 5\nkf = KFold(n_splits = K, random_state = 1, shuffle = True)\nnp.random.seed(0)","metadata":{"execution":{"iopub.status.busy":"2022-07-11T05:13:04.531705Z","iopub.execute_input":"2022-07-11T05:13:04.532140Z","iopub.status.idle":"2022-07-11T05:13:04.539904Z","shell.execute_reply.started":"2022-07-11T05:13:04.532106Z","shell.execute_reply":"2022-07-11T05:13:04.539108Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Set up classifier\nmodel = XGBClassifier(n_estimators = MAX_ROUNDS,\n                     max_depth = 4,\n                     objective = 'binary:logistic',\n                     learning_rate = LEARNING_RATE,\n                     subsample = .8,\n                     min_child_weight = 1.6,\n                     gamma = 10,\n                     reg_alpha = 8,\n                     reg_lambda = 1.3)","metadata":{"execution":{"iopub.status.busy":"2022-07-11T05:13:04.541498Z","iopub.execute_input":"2022-07-11T05:13:04.542389Z","iopub.status.idle":"2022-07-11T05:13:04.550210Z","shell.execute_reply.started":"2022-07-11T05:13:04.542343Z","shell.execute_reply":"2022-07-11T05:13:04.549207Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Run CV\n\nfor i, (train_index, test_index) in enumerate(kf.split(train_df)):\n    \n    # Create data for this fold\n    y_train, y_valid = y.iloc[train_index].copy(), y.iloc[test_index]\n    X_train, X_valid = X.iloc[train_index,:].copy(), X.iloc[test_index,:].copy()\n    X_test = test_df.copy()\n    print( \"\\nFold \", i)\n    \n    # Enocode data\n    for f in f_cats:\n        X_train[f + \"_avg\"], X_valid[f + \"_avg\"], X_test[f + \"_avg\"] = target_encode(\n                                                        trn_series=X_train[f],\n                                                        val_series=X_valid[f],\n                                                        tst_series=X_test[f],\n                                                        target=y_train,\n                                                        min_samples_leaf=200,\n                                                        smoothing=10,\n                                                        noise_level=0\n                                                        )\n    # Run model for this fold\n    if OPTIMIZE_ROUNDS:\n        eval_set=[(X_valid,y_valid)]\n        fit_model = model.fit( X_train, y_train, \n                               eval_set=eval_set,\n                               eval_metric=gini_xgb,\n                               early_stopping_rounds=EARLY_STOPPING_ROUNDS,\n                               verbose=False\n                             )\n        print( \"  Best N trees = \", model.best_ntree_limit )\n        print( \"  Best gini = \", model.best_score )\n    else:\n        fit_model = model.fit( X_train, y_train )\n        \n    # Generate validation predictions for this fold\n    pred = fit_model.predict_proba(X_valid)[:,1]\n    print( \"  Gini = \", eval_gini(y_valid, pred) )\n    y_valid_pred.iloc[test_index] = pred\n    \n    # Accumulate test set predictions\n    y_test_pred += fit_model.predict_proba(X_test)[:,1]\n    \n    del X_test, X_train, X_valid, y_train\n    \ny_test_pred /= K  # Average test set predictions\n\nprint( \"\\nGini for full training set:\" )\neval_gini(y, y_valid_pred)","metadata":{"execution":{"iopub.status.busy":"2022-07-11T05:13:04.552505Z","iopub.execute_input":"2022-07-11T05:13:04.552949Z","iopub.status.idle":"2022-07-11T05:30:35.862747Z","shell.execute_reply.started":"2022-07-11T05:13:04.552905Z","shell.execute_reply":"2022-07-11T05:30:35.861600Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Save validation predictions for stacking/ensembling\nval = pd.DataFrame()\nval['id'] = id_train\nval['target'] = y_valid_pred.values\nval.to_csv('./xgb_valid.csv', float_format = '%.6f', index = False)","metadata":{"execution":{"iopub.status.busy":"2022-07-11T05:36:20.391767Z","iopub.execute_input":"2022-07-11T05:36:20.392490Z","iopub.status.idle":"2022-07-11T05:36:22.990228Z","shell.execute_reply.started":"2022-07-11T05:36:20.392447Z","shell.execute_reply":"2022-07-11T05:36:22.988973Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create submission file\nsub = pd.DataFrame()\nsub['id'] = id_test\nsub['target'] = y_test_pred\nsub.to_csv('./xgb_submit.csv', float_format= '%.6f', index= False)","metadata":{"execution":{"iopub.status.busy":"2022-07-11T05:36:22.992601Z","iopub.execute_input":"2022-07-11T05:36:22.993063Z","iopub.status.idle":"2022-07-11T05:36:26.800676Z","shell.execute_reply.started":"2022-07-11T05:36:22.993004Z","shell.execute_reply":"2022-07-11T05:36:26.799296Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}