{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Packages","metadata":{}},{"cell_type":"code","source":"import time\nimport gc\nimport pickle\nimport warnings\n\nimport pandas as pd\nimport numpy as np\n\nfrom sklearn.metrics import confusion_matrix\nfrom sklearn.metrics import ConfusionMatrixDisplay\nfrom sklearn.model_selection import StratifiedKFold\n\nimport lightgbm as lgb\nfrom lightgbm import LGBMClassifier, log_evaluation, early_stopping","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-25T07:19:39.414286Z","iopub.execute_input":"2022-08-25T07:19:39.414696Z","iopub.status.idle":"2022-08-25T07:19:40.224967Z","shell.execute_reply.started":"2022-08-25T07:19:39.414608Z","shell.execute_reply":"2022-08-25T07:19:40.224119Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.options.display.max_rows = 300\npd.options.display.max_seq_items = 300\nDEBUG = True\nNROWS_DEBUG = 10000\nmodel_version=int(time.time())\ntrain_data_path = \"../input/amex-default-prediction-agg-data-preprocess/train_agg_data_1657174948.pkl\"\ntest_data_path = \"../input/amex-default-prediction-agg-data-preprocess/test_agg_data_1657174948.pkl\"\nprint(model_version)","metadata":{"execution":{"iopub.status.busy":"2022-08-25T07:19:40.226353Z","iopub.execute_input":"2022-08-25T07:19:40.226980Z","iopub.status.idle":"2022-08-25T07:19:40.234113Z","shell.execute_reply.started":"2022-08-25T07:19:40.226943Z","shell.execute_reply":"2022-08-25T07:19:40.233102Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load Data","metadata":{}},{"cell_type":"code","source":"%%time\nprint('Load data ...')\nif DEBUG:\n    train = pd.read_pickle(train_data_path, compression=\"gzip\")\n    train = train.head(NROWS_DEBUG)\n    test = pd.read_pickle(test_data_path, compression=\"gzip\")\n    test = test.head(NROWS_DEBUG)\nelse:\n    train = pd.read_pickle(train_data_path, compression=\"gzip\")\n    test = pd.read_pickle(test_data_path, compression=\"gzip\")","metadata":{"execution":{"iopub.status.busy":"2022-08-25T07:19:40.236178Z","iopub.execute_input":"2022-08-25T07:19:40.237004Z","iopub.status.idle":"2022-08-25T07:20:00.805266Z","shell.execute_reply.started":"2022-08-25T07:19:40.236976Z","shell.execute_reply":"2022-08-25T07:20:00.804137Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_y = pd.DataFrame(train['target'])\ntrain_x = train.drop('target', axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-08-25T07:20:00.807206Z","iopub.execute_input":"2022-08-25T07:20:00.807483Z","iopub.status.idle":"2022-08-25T07:20:00.865033Z","shell.execute_reply.started":"2022-08-25T07:20:00.807458Z","shell.execute_reply":"2022-08-25T07:20:00.863663Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Utility Functions","metadata":{}},{"cell_type":"markdown","source":"## AmEx Scoring Metric","metadata":{}},{"cell_type":"code","source":"# @yunchonggan's fast metric implementation\n# From https://www.kaggle.com/competitions/amex-default-prediction/discussion/328020\ndef amex_metric(y_true: np.array, y_pred: np.array) -> float:\n\n    # count of positives and negatives\n    n_pos = y_true.sum()\n    n_neg = y_true.shape[0] - n_pos\n\n    # sorting by descring prediction values\n    indices = np.argsort(y_pred)[::-1]\n    preds, target = y_pred[indices], y_true[indices]\n\n    # filter the top 4% by cumulative row weights\n    weight = 20.0 - target * 19.0\n    cum_norm_weight = (weight / weight.sum()).cumsum()\n    four_pct_filter = cum_norm_weight <= 0.04\n\n    # default rate captured at 4%\n    d = target[four_pct_filter].sum() / n_pos\n\n    # weighted gini coefficient\n    lorentz = (target / n_pos).cumsum()\n    gini = ((lorentz - cum_norm_weight) * weight).sum()\n\n    # max weighted gini coefficient\n    gini_max = 10 * n_neg * (1 - 19 / (n_pos + 20 * n_neg))\n\n    # normalized weighted gini coefficient\n    g = gini / gini_max\n\n    return 0.32 * np.log(g + d)\n\ndef lgb_amex_metric(y_true, y_pred):\n    \"\"\"The competition metric with lightgbm's calling convention\"\"\"\n    return ('amex',\n            amex_metric(y_true, y_pred),\n            True)","metadata":{"execution":{"iopub.status.busy":"2022-08-25T07:20:00.866138Z","iopub.execute_input":"2022-08-25T07:20:00.866406Z","iopub.status.idle":"2022-08-25T07:20:00.874898Z","shell.execute_reply.started":"2022-08-25T07:20:00.866383Z","shell.execute_reply":"2022-08-25T07:20:00.874102Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Simple Train","metadata":{}},{"cell_type":"markdown","source":"Train is actually `Parameters Tunning`\n\nWhat happend if we dont have train test split and implement Early Stopping ?\n\nTrain test split here to simulate test as unseen data","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nX_train, X_test, y_train, y_test = train_test_split(train_x, train_y, test_size=0.20, random_state=314, stratify=train_y)","metadata":{"execution":{"iopub.status.busy":"2022-08-25T07:20:02.263852Z","iopub.execute_input":"2022-08-25T07:20:02.264483Z","iopub.status.idle":"2022-08-25T07:20:02.379368Z","shell.execute_reply.started":"2022-08-25T07:20:02.264455Z","shell.execute_reply":"2022-08-25T07:20:02.377811Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tuned_params = {'colsample_bytree': 0.2, \n                'max_bins': 511, \n                'min_child_samples': 2400, \n                'num_leaves': 95, \n                'reg_lambda': 50, \n                } \ndef my_clf():\n    return LGBMClassifier(n_estimators=20000,  # numbers of tree\n                          learning_rate=0.02, \n                          random_state=42,\n                          **tuned_params\n                         )","metadata":{"execution":{"iopub.status.busy":"2022-08-25T07:20:09.944412Z","iopub.execute_input":"2022-08-25T07:20:09.944746Z","iopub.status.idle":"2022-08-25T07:20:09.950807Z","shell.execute_reply.started":"2022-08-25T07:20:09.944721Z","shell.execute_reply":"2022-08-25T07:20:09.949528Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## No Validation/Test Set and EarlyStopping","metadata":{}},{"cell_type":"markdown","source":"If we don't have the validation/test set as Early Stopping condition and keep on training, the loss on training on data will keep on decreasing. However, the loss of validation/test set will not keep on decreasing ","metadata":{}},{"cell_type":"code","source":"%%time\nclf = my_clf()\nclf.fit(X_train, \n        y_train.values.ravel(), \n        eval_set=[(X_train, y_train.values.ravel())],  # train set as evaluation set, you will see that the logloss keeps on decreasing\n        eval_metric=[lgb_amex_metric], \n        callbacks=[log_evaluation(500)]\n       )","metadata":{"execution":{"iopub.status.busy":"2022-08-25T07:20:15.643869Z","iopub.execute_input":"2022-08-25T07:20:15.644282Z","iopub.status.idle":"2022-08-25T07:21:41.307827Z","shell.execute_reply.started":"2022-08-25T07:20:15.644251Z","shell.execute_reply":"2022-08-25T07:21:41.306640Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Let's compare the loss between Train and Validation/Test Set","metadata":{}},{"cell_type":"code","source":"%%time\nclf = my_clf()\nclf.fit(X_train, \n        y_train.values.ravel(), \n        eval_set=[(X_test, y_test.values.ravel()), (X_train, y_train.values.ravel())], \n        eval_metric=[lgb_amex_metric],\n        callbacks=[log_evaluation(500)]\n       )","metadata":{"execution":{"iopub.status.busy":"2022-08-25T07:21:41.310106Z","iopub.execute_input":"2022-08-25T07:21:41.310529Z","iopub.status.idle":"2022-08-25T07:23:16.753875Z","shell.execute_reply.started":"2022-08-25T07:21:41.310494Z","shell.execute_reply":"2022-08-25T07:23:16.752931Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from matplotlib import pyplot as plt","metadata":{"execution":{"iopub.status.busy":"2022-08-25T07:23:16.754831Z","iopub.execute_input":"2022-08-25T07:23:16.755051Z","iopub.status.idle":"2022-08-25T07:23:16.759447Z","shell.execute_reply.started":"2022-08-25T07:23:16.755028Z","shell.execute_reply":"2022-08-25T07:23:16.758362Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.plot(clf.evals_result_['valid_0']['binary_logloss'])\nplt.plot(clf.evals_result_['valid_1']['binary_logloss'])\nplt.xlabel('Number of training steps (n_estimators in lgbm)')\nplt.ylabel('Loss (binary_logloss)')\nplt.legend(['test', 'train'])","metadata":{"execution":{"iopub.status.busy":"2022-08-25T07:23:16.762871Z","iopub.execute_input":"2022-08-25T07:23:16.763134Z","iopub.status.idle":"2022-08-25T07:23:16.990399Z","shell.execute_reply.started":"2022-08-25T07:23:16.763111Z","shell.execute_reply":"2022-08-25T07:23:16.988880Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Train Test Split and EarlyStopping","metadata":{}},{"cell_type":"markdown","source":"Recommended way is to have train or test set as Early Stopping condition","metadata":{}},{"cell_type":"code","source":"%%time\nclf = my_clf()\nclf.fit(X_train, \n        y_train.values.ravel(), \n        eval_set=[(X_test, y_test.values.ravel()), (X_train, y_train.values.ravel())], \n        eval_metric=[lgb_amex_metric], \n        callbacks=[log_evaluation(500), early_stopping(1000)]  # stop training if loss/metric doesn't change after 1000\n       )","metadata":{"execution":{"iopub.status.busy":"2022-08-25T07:23:16.991687Z","iopub.execute_input":"2022-08-25T07:23:16.991975Z","iopub.status.idle":"2022-08-25T07:23:39.792112Z","shell.execute_reply.started":"2022-08-25T07:23:16.991950Z","shell.execute_reply":"2022-08-25T07:23:39.790482Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"No `Hyperparameters` Tunning","metadata":{}},{"cell_type":"code","source":"plt.plot(clf.evals_result_['valid_0']['binary_logloss'])\nplt.plot(clf.evals_result_['valid_1']['binary_logloss'])\nplt.xlabel('Number of training steps (n_estimators in lgbm)')\nplt.ylabel('Loss (binary_logloss)')\nplt.legend(['test', 'train'])","metadata":{"execution":{"iopub.status.busy":"2022-08-25T07:23:39.793452Z","iopub.execute_input":"2022-08-25T07:23:39.793720Z","iopub.status.idle":"2022-08-25T07:23:39.963988Z","shell.execute_reply.started":"2022-08-25T07:23:39.793696Z","shell.execute_reply":"2022-08-25T07:23:39.963007Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Cross Validation","metadata":{}},{"cell_type":"markdown","source":"### Training with CV\nCross Validation (CV) is just the repeated process of training with validation set over N folds of the data","metadata":{}},{"cell_type":"code","source":"tuned_params = {'colsample_bytree': 0.2, \n                'max_bins': 511, \n                'min_child_samples': 2400, \n                'num_leaves': 95, \n                'reg_lambda': 50, \n                } \ndef my_clf():\n    return LGBMClassifier(n_estimators=6000,\n                          learning_rate=0.02, \n                          random_state=42,\n                          **tuned_params\n                         )","metadata":{"execution":{"iopub.status.busy":"2022-08-25T07:23:39.965089Z","iopub.execute_input":"2022-08-25T07:23:39.965307Z","iopub.status.idle":"2022-08-25T07:23:39.972535Z","shell.execute_reply.started":"2022-08-25T07:23:39.965283Z","shell.execute_reply":"2022-08-25T07:23:39.970830Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nN_FOLDS = 5  # split the data in to 5 folds\nskf = StratifiedKFold(n_splits=N_FOLDS, shuffle=True, random_state=22)\ny_oof = np.zeros(train_x.shape[0])\ny_test = np.zeros(test.shape[0])\nix = 0\nfor train_ind, val_ind in skf.split(train_x, train_y):\n    print(f\"******* Fold {ix} ******* \")\n    tr_x, val_x = (\n        train_x.iloc[train_ind].reset_index(drop=True),\n        train_x.iloc[val_ind].reset_index(drop=True),\n    )\n    tr_y, val_y = (\n        train_y.iloc[train_ind].reset_index(drop=True),\n        train_y.iloc[val_ind].reset_index(drop=True),\n    )\n\n    clf = my_clf()\n    clf.fit(tr_x, \n            tr_y.values.ravel(), \n            eval_set=[(val_x, val_y.values.ravel())], \n            eval_metric=[lgb_amex_metric], \n            callbacks=[log_evaluation(100), early_stopping(500)]\n           )\n    \n    preds = clf.predict_proba(val_x)[:, 1]\n    y_oof[val_ind] = preds\n\n    preds_test = clf.predict_proba(test)[:, 1]  # prediction for new here\n    y_test = y_test + preds_test / N_FOLDS\n    \n    # save model\n    with open('model_{}_fold_{}.pkl'.format(model_version, ix), 'wb') as outfile:\n        pickle.dump(clf, outfile)\n    \n    ix = ix + 1\n    \ny_pred = train_y.copy(deep=True)\ny_pred = y_pred.rename(columns={\"target\": \"prediction\"})\ny_pred[\"prediction\"] = y_oof\nval_score = amex_metric(train_y.target.values, y_pred.prediction.values)\nprint(f\"Amex metric: {val_score}\")\n\nwith open('y_oof_{}.pkl'.format(model_version), 'wb') as outfile:\n    pickle.dump(y_oof, outfile)\n\nwith open('y_test_{}.pkl'.format(model_version), 'wb') as outfile:\n    pickle.dump(y_test, outfile)","metadata":{"execution":{"iopub.status.busy":"2022-08-25T07:23:39.974318Z","iopub.execute_input":"2022-08-25T07:23:39.974810Z","iopub.status.idle":"2022-08-25T07:24:13.764004Z","shell.execute_reply.started":"2022-08-25T07:23:39.974773Z","shell.execute_reply":"2022-08-25T07:24:13.762208Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Confusion Matrix","metadata":{}},{"cell_type":"code","source":"y_oof_binary = (y_oof >= np.percentile(y_oof, 96)).astype(int)  # at top 4%\ny_oof_binary.mean()\nclass_names = [0,1]\n\n\ncf = confusion_matrix(train_y, y_oof_binary)\ndisp = ConfusionMatrixDisplay(confusion_matrix=cf, display_labels=class_names)\ndisp.plot()","metadata":{"execution":{"iopub.status.busy":"2022-08-25T07:24:13.765544Z","iopub.execute_input":"2022-08-25T07:24:13.765970Z","iopub.status.idle":"2022-08-25T07:24:13.942567Z","shell.execute_reply.started":"2022-08-25T07:24:13.765932Z","shell.execute_reply":"2022-08-25T07:24:13.941072Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Hyperparameters Tunning via RandomizedSearchCV","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nfrom sklearn.model_selection import RandomizedSearchCV\n\n\nfrom scipy.stats import randint as sp_randint\nfrom scipy.stats import uniform as sp_uniform\nfrom sklearn.metrics import make_scorer\namex_score = make_scorer(amex_metric, greater_is_better=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-25T07:41:58.666584Z","iopub.execute_input":"2022-08-25T07:41:58.666935Z","iopub.status.idle":"2022-08-25T07:41:58.672295Z","shell.execute_reply.started":"2022-08-25T07:41:58.666910Z","shell.execute_reply":"2022-08-25T07:41:58.671130Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Parameters to be searched","metadata":{}},{"cell_type":"code","source":"# param_test ={'num_leaves': sp_randint(20, 200), \n#              'min_child_samples': sp_randint(300, 5000), \n#              'min_child_weight': [1e-5, 1e-3, 1e-2, 1e-1, 1, 1e1, 1e2, 1e3, 1e4],\n#              'subsample': sp_uniform(loc=0.3, scale=0.8), \n#              'colsample_bytree': sp_uniform(loc=0.3, scale=0.8),\n#              'reg_alpha': [0, 1e-1, 1, 5, 10, 50, 100],\n#              'reg_lambda': [0, 1e-1, 1, 5, 10, 50, 100],\n#               'max_bins': [10, 100, 511]}\n\nparam_test ={'num_leaves': sp_randint(20, 200),  # 95\n             'min_child_samples': sp_randint(300, 5000),  # 2400\n#              'min_child_weight': [1e-5, 1e-3, 1e-2, 1e-1, 1, 1e1, 1e2, 1e3, 1e4],\n             'subsample': sp_uniform(loc=0.2, scale=0.8),\n             'colsample_bytree': sp_uniform(loc=0.1, scale=0.6),  # 0.19\n             'reg_lambda': [10, 50, 100],  # 50\n             'max_bins': [100, 511]  # 511\n            }","metadata":{"execution":{"iopub.status.busy":"2022-08-25T07:41:59.659197Z","iopub.execute_input":"2022-08-25T07:41:59.659847Z","iopub.status.idle":"2022-08-25T07:41:59.670757Z","shell.execute_reply.started":"2022-08-25T07:41:59.659812Z","shell.execute_reply":"2022-08-25T07:41:59.669592Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nX_train, X_test, y_train, y_test = train_test_split(train_x, train_y, test_size=0.20, random_state=314, stratify=train_y)","metadata":{"execution":{"iopub.status.busy":"2022-08-25T07:42:45.154599Z","iopub.execute_input":"2022-08-25T07:42:45.154961Z","iopub.status.idle":"2022-08-25T07:42:45.271767Z","shell.execute_reply.started":"2022-08-25T07:42:45.154935Z","shell.execute_reply":"2022-08-25T07:42:45.270616Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Fixed fit_params","metadata":{}},{"cell_type":"code","source":"if DEBUG:\n    n_HP_points_to_test = 10  # this parameter defines the number of HP points to be tested\n    cv=5 # cross-validation folds\n    fit_params={\"eval_metric\" : [lgb_amex_metric], \n                \"eval_set\" : [(X_test,y_test.values.ravel())],\n                'eval_names': ['valid'],\n                'callbacks': [log_evaluation(10), early_stopping(5)],\n                'categorical_feature': 'auto'}\nelse:\n    n_HP_points_to_test = 10\n    cv=5\n    fit_params={\"eval_metric\" : [lgb_amex_metric], \n                \"eval_set\" : [(X_test,y_test.values.ravel())],\n                'eval_names': ['valid'],\n                'callbacks': [log_evaluation(100), early_stopping(50)],\n                'categorical_feature': 'auto'}","metadata":{"execution":{"iopub.status.busy":"2022-08-25T07:42:45.769578Z","iopub.execute_input":"2022-08-25T07:42:45.769980Z","iopub.status.idle":"2022-08-25T07:42:45.777479Z","shell.execute_reply.started":"2022-08-25T07:42:45.769952Z","shell.execute_reply":"2022-08-25T07:42:45.776493Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"clf = LGBMClassifier(max_depth=-1, \n                     random_state=31, \n                     metric='None',\n                     learning_rate=0.03, \n                     n_jobs=4, \n                     n_estimators=5000\n                    )\n\ngs = RandomizedSearchCV(estimator=clf, \n                        param_distributions=param_test, \n                        n_iter=n_HP_points_to_test,\n                        scoring=amex_score,\n                        cv=cv,\n                        refit=True,\n                        random_state=314,\n                        verbose=100\n                        )","metadata":{"execution":{"iopub.status.busy":"2022-08-25T07:42:47.043962Z","iopub.execute_input":"2022-08-25T07:42:47.046168Z","iopub.status.idle":"2022-08-25T07:42:47.054307Z","shell.execute_reply.started":"2022-08-25T07:42:47.046127Z","shell.execute_reply":"2022-08-25T07:42:47.053217Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Search","metadata":{}},{"cell_type":"code","source":"%%time\ngs.fit(X_train, y_train.values.ravel(), **fit_params)\nprint('Best score reached: {} with params: {} '.format(gs.best_score_, gs.best_params_))","metadata":{"execution":{"iopub.status.busy":"2022-08-25T07:42:47.839388Z","iopub.execute_input":"2022-08-25T07:42:47.839910Z","iopub.status.idle":"2022-08-25T07:43:22.862866Z","shell.execute_reply.started":"2022-08-25T07:42:47.839884Z","shell.execute_reply":"2022-08-25T07:43:22.862190Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## After get best hyperparameters, retrain with (train-validation) and evaluate with `NEW` test data and early stopping","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}