{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## Configuration","metadata":{}},{"cell_type":"code","source":"import time\nimport gc\nimport pickle\nimport warnings\n\nimport pandas as pd\nimport numpy as np\n\nfrom sklearn.metrics import confusion_matrix\nfrom sklearn.metrics import ConfusionMatrixDisplay\nfrom sklearn.model_selection import StratifiedKFold\n\nimport xgboost as xgb","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-08T04:38:56.757951Z","iopub.execute_input":"2022-07-08T04:38:56.758795Z","iopub.status.idle":"2022-07-08T04:38:56.765224Z","shell.execute_reply.started":"2022-07-08T04:38:56.758753Z","shell.execute_reply":"2022-07-08T04:38:56.764035Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.options.display.max_rows = 300\npd.options.display.max_seq_items = 300\nDEBUG = False\nNROWS_DEBUG = 100000\nmodel_version=int(time.time())\ntrain_data_path = \"../input/amex-default-prediction-agg-data-preprocess/train_agg_data_1657174948.pkl\"\ntest_data_path = \"../input/amex-default-prediction-agg-data-preprocess/test_agg_data_1657174948.pkl\"\nprint(model_version)","metadata":{"execution":{"iopub.status.busy":"2022-07-08T04:38:57.548653Z","iopub.execute_input":"2022-07-08T04:38:57.549058Z","iopub.status.idle":"2022-07-08T04:38:57.555941Z","shell.execute_reply.started":"2022-07-08T04:38:57.549023Z","shell.execute_reply":"2022-07-08T04:38:57.554921Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Load data ...')","metadata":{"execution":{"iopub.status.busy":"2022-07-08T04:38:58.381197Z","iopub.execute_input":"2022-07-08T04:38:58.382221Z","iopub.status.idle":"2022-07-08T04:38:58.388753Z","shell.execute_reply.started":"2022-07-08T04:38:58.38217Z","shell.execute_reply":"2022-07-08T04:38:58.387476Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Load Data","metadata":{}},{"cell_type":"code","source":"%%time\nif DEBUG:\n    train = pd.read_pickle(train_data_path, compression=\"gzip\")\n    train = train.head(NROWS_DEBUG)\n    test = pd.read_pickle(test_data_path, compression=\"gzip\")\n    test = test.head(NROWS_DEBUG)\nelse:\n    train = pd.read_pickle(train_data_path, compression=\"gzip\")\n    test = pd.read_pickle(test_data_path, compression=\"gzip\")","metadata":{"execution":{"iopub.status.busy":"2022-07-08T04:39:01.565291Z","iopub.execute_input":"2022-07-08T04:39:01.565654Z","iopub.status.idle":"2022-07-08T04:39:26.222108Z","shell.execute_reply.started":"2022-07-08T04:39:01.565624Z","shell.execute_reply":"2022-07-08T04:39:26.220966Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_y = pd.DataFrame(train['target'])\ntrain_x = train.drop('target', axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-07-08T04:39:26.224291Z","iopub.execute_input":"2022-07-08T04:39:26.225004Z","iopub.status.idle":"2022-07-08T04:39:26.684669Z","shell.execute_reply.started":"2022-07-08T04:39:26.224959Z","shell.execute_reply":"2022-07-08T04:39:26.683672Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Utility Functions","metadata":{}},{"cell_type":"code","source":"# @yunchonggan's fast metric implementation\n# From https://www.kaggle.com/competitions/amex-default-prediction/discussion/328020\ndef amex_metric(y_true: np.array, y_pred: np.array) -> float:\n\n    # count of positives and negatives\n    n_pos = y_true.sum()\n    n_neg = y_true.shape[0] - n_pos\n\n    # sorting by descring prediction values\n    indices = np.argsort(y_pred)[::-1]\n    preds, target = y_pred[indices], y_true[indices]\n\n    # filter the top 4% by cumulative row weights\n    weight = 20.0 - target * 19.0\n    cum_norm_weight = (weight / weight.sum()).cumsum()\n    four_pct_filter = cum_norm_weight <= 0.04\n\n    # default rate captured at 4%\n    d = target[four_pct_filter].sum() / n_pos\n\n    # weighted gini coefficient\n    lorentz = (target / n_pos).cumsum()\n    gini = ((lorentz - cum_norm_weight) * weight).sum()\n\n    # max weighted gini coefficient\n    gini_max = 10 * n_neg * (1 - 19 / (n_pos + 20 * n_neg))\n\n    # normalized weighted gini coefficient\n    g = gini / gini_max\n\n    return 0.5 * (g + d)\n\n\ndef xgb_amex_metric(y_true, y_pred):\n    return -1*amex_metric(y_true, y_pred)","metadata":{"execution":{"iopub.status.busy":"2022-07-08T04:39:26.68643Z","iopub.execute_input":"2022-07-08T04:39:26.687084Z","iopub.status.idle":"2022-07-08T04:39:26.696794Z","shell.execute_reply.started":"2022-07-08T04:39:26.687042Z","shell.execute_reply":"2022-07-08T04:39:26.695883Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## XGB Model RandomizedSearchCV","metadata":{}},{"cell_type":"code","source":"# from sklearn.model_selection import train_test_split\n# from sklearn.model_selection import RandomizedSearchCV\n\n\n# from scipy.stats import randint as sp_randint\n# from scipy.stats import uniform as sp_uniform\n# from sklearn.metrics import make_scorer\n# amex_score = make_scorer(amex_metric, greater_is_better=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-08T04:39:26.699343Z","iopub.execute_input":"2022-07-08T04:39:26.701055Z","iopub.status.idle":"2022-07-08T04:39:26.71052Z","shell.execute_reply.started":"2022-07-08T04:39:26.701028Z","shell.execute_reply":"2022-07-08T04:39:26.709602Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# print('Prepare RandomizedSearchCV model...')","metadata":{"execution":{"iopub.status.busy":"2022-07-08T04:39:26.712303Z","iopub.execute_input":"2022-07-08T04:39:26.713046Z","iopub.status.idle":"2022-07-08T04:39:26.721355Z","shell.execute_reply.started":"2022-07-08T04:39:26.71301Z","shell.execute_reply":"2022-07-08T04:39:26.720435Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# %%time\n# X_train, X_test, y_train, y_test = train_test_split(train_x, train_y, test_size=0.20, random_state=314, stratify=train_y)","metadata":{"execution":{"iopub.status.busy":"2022-07-08T04:39:26.724625Z","iopub.execute_input":"2022-07-08T04:39:26.725357Z","iopub.status.idle":"2022-07-08T04:39:26.730843Z","shell.execute_reply.started":"2022-07-08T04:39:26.725318Z","shell.execute_reply":"2022-07-08T04:39:26.729797Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Parameters to be searched","metadata":{}},{"cell_type":"code","source":"# # param_test ={'num_leaves': sp_randint(20, 200), \n# #              'min_child_samples': sp_randint(300, 5000), \n# #              'min_child_weight': [1e-5, 1e-3, 1e-2, 1e-1, 1, 1e1, 1e2, 1e3, 1e4],\n# #              'subsample': sp_uniform(loc=0.3, scale=0.8), \n# #              'colsample_bytree': sp_uniform(loc=0.3, scale=0.8),\n# #              'reg_alpha': [0, 1e-1, 1, 5, 10, 50, 100],\n# #              'reg_lambda': [0, 1e-1, 1, 5, 10, 50, 100],\n# #               'max_bins': [10, 100, 511]}\n\n# param_test ={'num_leaves': sp_randint(20, 200),  # 95\n#              'min_child_samples': sp_randint(300, 5000),  # 2400\n# #              'min_child_weight': [1e-5, 1e-3, 1e-2, 1e-1, 1, 1e1, 1e2, 1e3, 1e4],\n#              'subsample': sp_uniform(loc=0.2, scale=0.8),\n#              'colsample_bytree': sp_uniform(loc=0.1, scale=0.6),  # 0.19\n#              'reg_lambda': [10, 50, 100],  # 50\n#              'max_bins': [100, 511]  # 511\n#             }\n\n# #     return LGBMClassifier(n_estimators=n_estimators,\n# #                           learning_rate=0.03, reg_lambda=50,\n# #                           min_child_samples=2400,\n# #                           num_leaves=95,\n# #                           colsample_bytree=0.19,\n# #                           max_bins=511, random_state=random_state)","metadata":{"execution":{"iopub.status.busy":"2022-07-08T04:39:26.732773Z","iopub.execute_input":"2022-07-08T04:39:26.733198Z","iopub.status.idle":"2022-07-08T04:39:26.741167Z","shell.execute_reply.started":"2022-07-08T04:39:26.733161Z","shell.execute_reply":"2022-07-08T04:39:26.74022Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Fixed fit_params","metadata":{}},{"cell_type":"code","source":"# if DEBUG:\n#     n_HP_points_to_test = 2  # this parameter defines the number of HP points to be tested\n#     cv=2  # cross-validation folds\n#     fit_params={\"eval_metric\" : [lgb_amex_metric], \n#                 \"eval_set\" : [(X_test,y_test.values.ravel())],\n#                 'eval_names': ['valid'],\n#                 'callbacks': [log_evaluation(10), early_stopping(5)],\n#                 'categorical_feature': 'auto'}\n# else:\n#     n_HP_points_to_test = 10\n#     cv=5\n#     fit_params={\"eval_metric\" : [lgb_amex_metric], \n#                 \"eval_set\" : [(X_test,y_test.values.ravel())],\n#                 'eval_names': ['valid'],\n#                 'callbacks': [log_evaluation(100), early_stopping(50)],\n#                 'categorical_feature': 'auto'}","metadata":{"execution":{"iopub.status.busy":"2022-07-08T04:39:26.743015Z","iopub.execute_input":"2022-07-08T04:39:26.743694Z","iopub.status.idle":"2022-07-08T04:39:26.753955Z","shell.execute_reply.started":"2022-07-08T04:39:26.743658Z","shell.execute_reply":"2022-07-08T04:39:26.75291Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# clf = LGBMClassifier(max_depth=-1, \n#                      random_state=31, \n#                      metric='None',\n#                      learning_rate=0.03, \n#                      n_jobs=4, \n#                      n_estimators=5000\n#                     )\n\n# gs = RandomizedSearchCV(estimator=clf, \n#                         param_distributions=param_test, \n#                         n_iter=n_HP_points_to_test,\n#                         scoring=amex_score,\n#                         cv=cv,\n#                         refit=True,\n#                         random_state=314,\n#                         verbose=100\n#                         )","metadata":{"execution":{"iopub.status.busy":"2022-07-08T04:39:26.755256Z","iopub.execute_input":"2022-07-08T04:39:26.756109Z","iopub.status.idle":"2022-07-08T04:39:26.773825Z","shell.execute_reply.started":"2022-07-08T04:39:26.756073Z","shell.execute_reply":"2022-07-08T04:39:26.772936Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Search","metadata":{}},{"cell_type":"code","source":"# print('RandomizedSearchCV model...')","metadata":{"execution":{"iopub.status.busy":"2022-07-08T04:39:26.778213Z","iopub.execute_input":"2022-07-08T04:39:26.778517Z","iopub.status.idle":"2022-07-08T04:39:26.783846Z","shell.execute_reply.started":"2022-07-08T04:39:26.778491Z","shell.execute_reply":"2022-07-08T04:39:26.78274Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# %%time\n# gs.fit(X_train, y_train.values.ravel(), **fit_params)\n# print('Best score reached: {} with params: {} '.format(gs.best_score_, gs.best_params_))","metadata":{"execution":{"iopub.status.busy":"2022-07-08T04:39:26.785607Z","iopub.execute_input":"2022-07-08T04:39:26.78605Z","iopub.status.idle":"2022-07-08T04:39:26.792778Z","shell.execute_reply.started":"2022-07-08T04:39:26.785947Z","shell.execute_reply":"2022-07-08T04:39:26.791939Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Saving and Submission","metadata":{}},{"cell_type":"code","source":"# print('Saving grid search result...')","metadata":{"execution":{"iopub.status.busy":"2022-07-08T04:39:26.794862Z","iopub.execute_input":"2022-07-08T04:39:26.795144Z","iopub.status.idle":"2022-07-08T04:39:26.802533Z","shell.execute_reply.started":"2022-07-08T04:39:26.79512Z","shell.execute_reply":"2022-07-08T04:39:26.801477Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# %%time\n# with open('gs_model_{}.pkl'.format(model_version), 'wb') as outfile:\n#     pickle.dump(gs, outfile)\n    \n# clf = gs.best_estimator_\n\n# y_pred = train_y.copy(deep=True)\n# y_pred = y_pred.rename(columns={\"target\": \"prediction\"})\n# y_pred[\"prediction\"] = clf.predict_proba(train_x)[:, 1]\n# val_score = amex_metric(train_y.target.values, y_pred.prediction.values)\n# print(f\"Amex metric: {val_score}\")\n\n# y_test = clf.predict_proba(test)[:, 1]\n# with open('y_test_{}.pkl'.format(model_version), 'wb') as outfile:\n#     pickle.dump(y_test, outfile)\n\n# test['prediction'] = y_test\n# test['prediction'].to_csv('submission_model_{}_val_{}.csv'.format(model_version, val_score), index=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-08T04:39:26.803841Z","iopub.execute_input":"2022-07-08T04:39:26.805433Z","iopub.status.idle":"2022-07-08T04:39:26.812481Z","shell.execute_reply.started":"2022-07-08T04:39:26.805338Z","shell.execute_reply":"2022-07-08T04:39:26.811416Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Tuned XGB Model","metadata":{}},{"cell_type":"markdown","source":"### Tuned Parameters","metadata":{}},{"cell_type":"code","source":"tuned_params = {'max_depth':4, \n                'subsample':0.8,\n                'colsample_bytree':0.6, \n                'eval_metric': xgb_amex_metric,\n                'objective':'binary:logistic',\n#                 'tree_method':'gpu_hist',\n#                 'predictor':'gpu_predictor',\n                } \ndef my_clf():\n    return xgb.XGBClassifier(n_estimators=5000,\n                             learning_rate=0.05, \n                             random_state=42,\n                             early_stopping_rounds=500,\n                             **tuned_params)","metadata":{"execution":{"iopub.status.busy":"2022-07-08T04:39:26.813672Z","iopub.execute_input":"2022-07-08T04:39:26.816251Z","iopub.status.idle":"2022-07-08T04:39:26.824254Z","shell.execute_reply.started":"2022-07-08T04:39:26.816223Z","shell.execute_reply":"2022-07-08T04:39:26.823268Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Training with CV","metadata":{}},{"cell_type":"code","source":"print('Training model ...')","metadata":{"execution":{"iopub.status.busy":"2022-07-08T04:39:26.827054Z","iopub.execute_input":"2022-07-08T04:39:26.828038Z","iopub.status.idle":"2022-07-08T04:39:26.83497Z","shell.execute_reply.started":"2022-07-08T04:39:26.828Z","shell.execute_reply":"2022-07-08T04:39:26.833844Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nN_FOLDS = 5\nskf = StratifiedKFold(n_splits=N_FOLDS, shuffle=True, random_state=22)\ny_oof = np.zeros(train_x.shape[0])\ny_test = np.zeros(test.shape[0])\nix = 0\nfor train_ind, val_ind in skf.split(train_x, train_y):\n    print(f\"******* Fold {ix} ******* \")\n    tr_x, val_x = (\n        train_x.iloc[train_ind].reset_index(drop=True),\n        train_x.iloc[val_ind].reset_index(drop=True),\n    )\n    tr_y, val_y = (\n        train_y.iloc[train_ind].reset_index(drop=True),\n        train_y.iloc[val_ind].reset_index(drop=True),\n    )\n   \n    clf = my_clf()\n    clf.fit(tr_x, \n            tr_y.values.ravel(), \n            eval_set=[(val_x, val_y.values.ravel())],\n            verbose=100\n           )\n    preds = clf.predict_proba(val_x)[:, 1]\n    y_oof[val_ind] = y_oof[val_ind] + preds\n\n    preds_test = clf.predict_proba(test)[:, 1]\n    y_test = y_test + preds_test / N_FOLDS\n    \n    # save model\n    with open('model_{}_fold_{}.pkl'.format(model_version, ix), 'wb') as outfile:\n        pickle.dump(clf, outfile)\n    \n    ix = ix + 1\n    \ny_pred = train_y.copy(deep=True)\ny_pred = y_pred.rename(columns={\"target\": \"prediction\"})\ny_pred[\"prediction\"] = y_oof\nval_score = amex_metric(train_y.target.values, y_pred.prediction.values)\nprint(f\"Amex metric: {val_score}\")\n\nwith open('y_oof_{}.pkl'.format(model_version), 'wb') as outfile:\n    pickle.dump(y_oof, outfile)\n\nwith open('y_test_{}.pkl'.format(model_version), 'wb') as outfile:\n    pickle.dump(y_test, outfile)","metadata":{"execution":{"iopub.status.busy":"2022-07-08T04:39:26.836411Z","iopub.execute_input":"2022-07-08T04:39:26.836881Z","iopub.status.idle":"2022-07-08T04:41:55.078534Z","shell.execute_reply.started":"2022-07-08T04:39:26.836835Z","shell.execute_reply":"2022-07-08T04:41:55.07755Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Submission","metadata":{}},{"cell_type":"code","source":"test['prediction'] = y_test\ntest['prediction'].to_csv('submission_model_{}_val_{}.csv'.format(model_version, val_score), index=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-08T04:42:00.483206Z","iopub.execute_input":"2022-07-08T04:42:00.48392Z","iopub.status.idle":"2022-07-08T04:42:00.998461Z","shell.execute_reply.started":"2022-07-08T04:42:00.483864Z","shell.execute_reply":"2022-07-08T04:42:00.997412Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Confusion Matrix","metadata":{}},{"cell_type":"code","source":"y_oof_binary = (y_oof >= np.percentile(y_oof, 96)).astype(int)  # at top 4%\ny_oof_binary.mean()","metadata":{"execution":{"iopub.status.busy":"2022-07-08T04:42:03.71359Z","iopub.execute_input":"2022-07-08T04:42:03.713968Z","iopub.status.idle":"2022-07-08T04:42:03.724138Z","shell.execute_reply.started":"2022-07-08T04:42:03.713934Z","shell.execute_reply":"2022-07-08T04:42:03.723078Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class_names = [0,1]\ncf = confusion_matrix(train_y, y_oof_binary)\ndisp = ConfusionMatrixDisplay(confusion_matrix=cf, display_labels=class_names)\ndisp.plot()","metadata":{"execution":{"iopub.status.busy":"2022-07-08T04:42:04.109614Z","iopub.execute_input":"2022-07-08T04:42:04.10988Z","iopub.status.idle":"2022-07-08T04:42:04.319695Z","shell.execute_reply.started":"2022-07-08T04:42:04.109855Z","shell.execute_reply":"2022-07-08T04:42:04.31878Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}