{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## Configuration","metadata":{}},{"cell_type":"code","source":"import time\nimport gc\nimport pickle\nimport warnings\n\nimport pandas as pd\nimport numpy as np\n\nfrom sklearn.metrics import confusion_matrix\nfrom sklearn.metrics import ConfusionMatrixDisplay\nfrom sklearn.model_selection import StratifiedKFold\n\nimport lightgbm as lgb\nfrom lightgbm import LGBMClassifier, log_evaluation, early_stopping","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-08T03:21:57.003644Z","iopub.execute_input":"2022-07-08T03:21:57.004019Z","iopub.status.idle":"2022-07-08T03:21:58.400099Z","shell.execute_reply.started":"2022-07-08T03:21:57.00393Z","shell.execute_reply":"2022-07-08T03:21:58.398764Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.options.display.max_rows = 300\npd.options.display.max_seq_items = 300\nDEBUG = False\nNROWS_DEBUG = 10000\nmodel_version=int(time.time())\ntrain_data_path = \"../input/amex-default-prediction-agg-data-preprocess/train_agg_data_1657174948.pkl\"\ntest_data_path = \"../input/amex-default-prediction-agg-data-preprocess/test_agg_data_1657174948.pkl\"\nprint(model_version)","metadata":{"execution":{"iopub.status.busy":"2022-07-08T03:21:58.402063Z","iopub.execute_input":"2022-07-08T03:21:58.402534Z","iopub.status.idle":"2022-07-08T03:21:58.410678Z","shell.execute_reply.started":"2022-07-08T03:21:58.402491Z","shell.execute_reply":"2022-07-08T03:21:58.409531Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Load data ...')","metadata":{"execution":{"iopub.status.busy":"2022-07-08T03:21:58.411641Z","iopub.execute_input":"2022-07-08T03:21:58.41195Z","iopub.status.idle":"2022-07-08T03:21:58.420581Z","shell.execute_reply.started":"2022-07-08T03:21:58.411922Z","shell.execute_reply":"2022-07-08T03:21:58.419442Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Load Data","metadata":{}},{"cell_type":"code","source":"%%time\nif DEBUG:\n    train = pd.read_pickle(train_data_path, compression=\"gzip\")\n    train = train.head(NROWS_DEBUG)\n    test = pd.read_pickle(test_data_path, compression=\"gzip\")\n    test = test.head(NROWS_DEBUG)\nelse:\n    train = pd.read_pickle(train_data_path, compression=\"gzip\")\n    test = pd.read_pickle(test_data_path, compression=\"gzip\")","metadata":{"execution":{"iopub.status.busy":"2022-07-08T03:21:58.425312Z","iopub.execute_input":"2022-07-08T03:21:58.425633Z","iopub.status.idle":"2022-07-08T03:22:27.078822Z","shell.execute_reply.started":"2022-07-08T03:21:58.425605Z","shell.execute_reply":"2022-07-08T03:22:27.07772Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_y = pd.DataFrame(train['target'])\ntrain_x = train.drop('target', axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-07-08T03:22:27.079987Z","iopub.execute_input":"2022-07-08T03:22:27.080301Z","iopub.status.idle":"2022-07-08T03:22:27.151794Z","shell.execute_reply.started":"2022-07-08T03:22:27.080272Z","shell.execute_reply":"2022-07-08T03:22:27.150723Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Utility Functions","metadata":{}},{"cell_type":"code","source":"# @yunchonggan's fast metric implementation\n# From https://www.kaggle.com/competitions/amex-default-prediction/discussion/328020\ndef amex_metric(y_true: np.array, y_pred: np.array) -> float:\n\n    # count of positives and negatives\n    n_pos = y_true.sum()\n    n_neg = y_true.shape[0] - n_pos\n\n    # sorting by descring prediction values\n    indices = np.argsort(y_pred)[::-1]\n    preds, target = y_pred[indices], y_true[indices]\n\n    # filter the top 4% by cumulative row weights\n    weight = 20.0 - target * 19.0\n    cum_norm_weight = (weight / weight.sum()).cumsum()\n    four_pct_filter = cum_norm_weight <= 0.04\n\n    # default rate captured at 4%\n    d = target[four_pct_filter].sum() / n_pos\n\n    # weighted gini coefficient\n    lorentz = (target / n_pos).cumsum()\n    gini = ((lorentz - cum_norm_weight) * weight).sum()\n\n    # max weighted gini coefficient\n    gini_max = 10 * n_neg * (1 - 19 / (n_pos + 20 * n_neg))\n\n    # normalized weighted gini coefficient\n    g = gini / gini_max\n\n    return 0.5 * (g + d)\n\ndef lgb_amex_metric(y_true, y_pred):\n    \"\"\"The competition metric with lightgbm's calling convention\"\"\"\n    return ('amex',\n            amex_metric(y_true, y_pred),\n            True)","metadata":{"execution":{"iopub.status.busy":"2022-07-08T03:22:27.153033Z","iopub.execute_input":"2022-07-08T03:22:27.15341Z","iopub.status.idle":"2022-07-08T03:22:27.165547Z","shell.execute_reply.started":"2022-07-08T03:22:27.153377Z","shell.execute_reply":"2022-07-08T03:22:27.164245Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## LightGBM Model RandomizedSearchCV","metadata":{}},{"cell_type":"code","source":"# from sklearn.model_selection import train_test_split\n# from sklearn.model_selection import RandomizedSearchCV\n\n\n# from scipy.stats import randint as sp_randint\n# from scipy.stats import uniform as sp_uniform\n# from sklearn.metrics import make_scorer\n# amex_score = make_scorer(amex_metric, greater_is_better=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-08T03:22:27.167399Z","iopub.execute_input":"2022-07-08T03:22:27.168474Z","iopub.status.idle":"2022-07-08T03:22:27.18097Z","shell.execute_reply.started":"2022-07-08T03:22:27.168422Z","shell.execute_reply":"2022-07-08T03:22:27.179734Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# print('Prepare RandomizedSearchCV model...')","metadata":{"execution":{"iopub.status.busy":"2022-07-08T03:22:27.182471Z","iopub.execute_input":"2022-07-08T03:22:27.183284Z","iopub.status.idle":"2022-07-08T03:22:27.19145Z","shell.execute_reply.started":"2022-07-08T03:22:27.183241Z","shell.execute_reply":"2022-07-08T03:22:27.190291Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# %%time\n# X_train, X_test, y_train, y_test = train_test_split(train_x, train_y, test_size=0.20, random_state=314, stratify=train_y)","metadata":{"execution":{"iopub.status.busy":"2022-07-08T03:22:27.193687Z","iopub.execute_input":"2022-07-08T03:22:27.194743Z","iopub.status.idle":"2022-07-08T03:22:27.200906Z","shell.execute_reply.started":"2022-07-08T03:22:27.194667Z","shell.execute_reply":"2022-07-08T03:22:27.19976Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Parameters to be searched","metadata":{}},{"cell_type":"code","source":"# # param_test ={'num_leaves': sp_randint(20, 200), \n# #              'min_child_samples': sp_randint(300, 5000), \n# #              'min_child_weight': [1e-5, 1e-3, 1e-2, 1e-1, 1, 1e1, 1e2, 1e3, 1e4],\n# #              'subsample': sp_uniform(loc=0.3, scale=0.8), \n# #              'colsample_bytree': sp_uniform(loc=0.3, scale=0.8),\n# #              'reg_alpha': [0, 1e-1, 1, 5, 10, 50, 100],\n# #              'reg_lambda': [0, 1e-1, 1, 5, 10, 50, 100],\n# #               'max_bins': [10, 100, 511]}\n\n# param_test ={'num_leaves': sp_randint(20, 200),  # 95\n#              'min_child_samples': sp_randint(300, 5000),  # 2400\n# #              'min_child_weight': [1e-5, 1e-3, 1e-2, 1e-1, 1, 1e1, 1e2, 1e3, 1e4],\n#              'subsample': sp_uniform(loc=0.2, scale=0.8),\n#              'colsample_bytree': sp_uniform(loc=0.1, scale=0.6),  # 0.19\n#              'reg_lambda': [10, 50, 100],  # 50\n#              'max_bins': [100, 511]  # 511\n#             }\n\n# #     return LGBMClassifier(n_estimators=n_estimators,\n# #                           learning_rate=0.03, reg_lambda=50,\n# #                           min_child_samples=2400,\n# #                           num_leaves=95,\n# #                           colsample_bytree=0.19,\n# #                           max_bins=511, random_state=random_state)","metadata":{"execution":{"iopub.status.busy":"2022-07-08T03:22:27.202116Z","iopub.execute_input":"2022-07-08T03:22:27.202924Z","iopub.status.idle":"2022-07-08T03:22:27.212753Z","shell.execute_reply.started":"2022-07-08T03:22:27.20289Z","shell.execute_reply":"2022-07-08T03:22:27.211611Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Fixed fit_params","metadata":{}},{"cell_type":"code","source":"# if DEBUG:\n#     n_HP_points_to_test = 2  # this parameter defines the number of HP points to be tested\n#     cv=2  # cross-validation folds\n#     fit_params={\"eval_metric\" : [lgb_amex_metric], \n#                 \"eval_set\" : [(X_test,y_test.values.ravel())],\n#                 'eval_names': ['valid'],\n#                 'callbacks': [log_evaluation(10), early_stopping(5)],\n#                 'categorical_feature': 'auto'}\n# else:\n#     n_HP_points_to_test = 10\n#     cv=5\n#     fit_params={\"eval_metric\" : [lgb_amex_metric], \n#                 \"eval_set\" : [(X_test,y_test.values.ravel())],\n#                 'eval_names': ['valid'],\n#                 'callbacks': [log_evaluation(100), early_stopping(50)],\n#                 'categorical_feature': 'auto'}","metadata":{"execution":{"iopub.status.busy":"2022-07-08T03:22:27.214366Z","iopub.execute_input":"2022-07-08T03:22:27.215027Z","iopub.status.idle":"2022-07-08T03:22:27.227126Z","shell.execute_reply.started":"2022-07-08T03:22:27.21499Z","shell.execute_reply":"2022-07-08T03:22:27.225886Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# clf = LGBMClassifier(max_depth=-1, \n#                      random_state=31, \n#                      metric='None',\n#                      learning_rate=0.03, \n#                      n_jobs=4, \n#                      n_estimators=5000\n#                     )\n\n# gs = RandomizedSearchCV(estimator=clf, \n#                         param_distributions=param_test, \n#                         n_iter=n_HP_points_to_test,\n#                         scoring=amex_score,\n#                         cv=cv,\n#                         refit=True,\n#                         random_state=314,\n#                         verbose=100\n#                         )","metadata":{"execution":{"iopub.status.busy":"2022-07-08T03:22:27.233423Z","iopub.execute_input":"2022-07-08T03:22:27.234579Z","iopub.status.idle":"2022-07-08T03:22:27.240463Z","shell.execute_reply.started":"2022-07-08T03:22:27.234521Z","shell.execute_reply":"2022-07-08T03:22:27.23939Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Search","metadata":{}},{"cell_type":"code","source":"# print('RandomizedSearchCV model...')","metadata":{"execution":{"iopub.status.busy":"2022-07-08T03:22:27.24205Z","iopub.execute_input":"2022-07-08T03:22:27.243159Z","iopub.status.idle":"2022-07-08T03:22:27.250384Z","shell.execute_reply.started":"2022-07-08T03:22:27.243119Z","shell.execute_reply":"2022-07-08T03:22:27.249254Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# %%time\n# gs.fit(X_train, y_train.values.ravel(), **fit_params)\n# print('Best score reached: {} with params: {} '.format(gs.best_score_, gs.best_params_))","metadata":{"execution":{"iopub.status.busy":"2022-07-08T03:22:27.252276Z","iopub.execute_input":"2022-07-08T03:22:27.253037Z","iopub.status.idle":"2022-07-08T03:22:27.267737Z","shell.execute_reply.started":"2022-07-08T03:22:27.25299Z","shell.execute_reply":"2022-07-08T03:22:27.266302Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Saving and Submission","metadata":{}},{"cell_type":"code","source":"# print('Saving grid search result...')","metadata":{"execution":{"iopub.status.busy":"2022-07-08T03:22:27.269099Z","iopub.execute_input":"2022-07-08T03:22:27.270075Z","iopub.status.idle":"2022-07-08T03:22:27.274711Z","shell.execute_reply.started":"2022-07-08T03:22:27.270018Z","shell.execute_reply":"2022-07-08T03:22:27.273705Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# %%time\n# with open('gs_model_{}.pkl'.format(model_version), 'wb') as outfile:\n#     pickle.dump(gs, outfile)\n    \n# clf = gs.best_estimator_\n\n# y_pred = train_y.copy(deep=True)\n# y_pred = y_pred.rename(columns={\"target\": \"prediction\"})\n# y_pred[\"prediction\"] = clf.predict_proba(train_x)[:, 1]\n# val_score = amex_metric(train_y.target.values, y_pred.prediction.values)\n# print(f\"Amex metric: {val_score}\")\n\n# y_test = clf.predict_proba(test)[:, 1]\n# with open('y_test_{}.pkl'.format(model_version), 'wb') as outfile:\n#     pickle.dump(y_test, outfile)\n\n# test['prediction'] = y_test\n# test['prediction'].to_csv('submission_model_{}_val_{}.csv'.format(model_version, val_score), index=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-08T03:22:27.276067Z","iopub.execute_input":"2022-07-08T03:22:27.277183Z","iopub.status.idle":"2022-07-08T03:22:27.286599Z","shell.execute_reply.started":"2022-07-08T03:22:27.277134Z","shell.execute_reply":"2022-07-08T03:22:27.285499Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Tuned LightGBM Model","metadata":{}},{"cell_type":"markdown","source":"### Tuned Parameters","metadata":{}},{"cell_type":"code","source":"# tuned_params = {'colsample_bytree': 0.42112048049033324, \n#                 'max_bins': 511, \n#                 'min_child_samples': 798, \n#                 'num_leaves': 148, \n#                 'reg_lambda': 10, \n#                 'subsample': 0.3215813935282683\n#                 } \ntuned_params = {'colsample_bytree': 0.2, \n                'max_bins': 511, \n                'min_child_samples': 2400, \n                'num_leaves': 95, \n                'reg_lambda': 50, \n                } \ndef my_clf():\n#     return LGBMClassifier(n_estimators=5000,\n#                           learning_rate=0.03, \n#                           reg_lambda=50,\n#                           min_child_samples=2400,\n#                           num_leaves=95,\n#                           colsample_bytree=0.19,\n#                           max_bins=511,\n#                           random_state=42\n#                          )\n\n    return LGBMClassifier(n_estimators=6000,\n                          learning_rate=0.02, \n                          random_state=42,\n                          **tuned_params\n                         )","metadata":{"execution":{"iopub.status.busy":"2022-07-08T03:22:27.288046Z","iopub.execute_input":"2022-07-08T03:22:27.288868Z","iopub.status.idle":"2022-07-08T03:22:27.3018Z","shell.execute_reply.started":"2022-07-08T03:22:27.288822Z","shell.execute_reply":"2022-07-08T03:22:27.300569Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Training with CV","metadata":{}},{"cell_type":"code","source":"print('Training model ...')","metadata":{"execution":{"iopub.status.busy":"2022-07-08T03:22:27.303634Z","iopub.execute_input":"2022-07-08T03:22:27.304603Z","iopub.status.idle":"2022-07-08T03:22:27.312805Z","shell.execute_reply.started":"2022-07-08T03:22:27.304549Z","shell.execute_reply":"2022-07-08T03:22:27.311638Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nN_FOLDS = 5\nskf = StratifiedKFold(n_splits=N_FOLDS, shuffle=True, random_state=22)\ny_oof = np.zeros(train_x.shape[0])\ny_test = np.zeros(test.shape[0])\nix = 0\nfor train_ind, val_ind in skf.split(train_x, train_y):\n    print(f\"******* Fold {ix} ******* \")\n    tr_x, val_x = (\n        train_x.iloc[train_ind].reset_index(drop=True),\n        train_x.iloc[val_ind].reset_index(drop=True),\n    )\n    tr_y, val_y = (\n        train_y.iloc[train_ind].reset_index(drop=True),\n        train_y.iloc[val_ind].reset_index(drop=True),\n    )\n\n    clf = my_clf()\n    clf.fit(tr_x, \n            tr_y.values.ravel(), \n            eval_set=[(val_x, val_y.values.ravel())], \n            eval_metric=[lgb_amex_metric], \n            callbacks=[log_evaluation(100), early_stopping(500)]\n           )\n    \n    preds = clf.predict_proba(val_x)[:, 1]\n    y_oof[val_ind] = y_oof[val_ind] + preds\n\n    preds_test = clf.predict_proba(test)[:, 1]\n    y_test = y_test + preds_test / N_FOLDS\n    \n    # save model\n    with open('model_{}_fold_{}.pkl'.format(model_version, ix), 'wb') as outfile:\n        pickle.dump(clf, outfile)\n    \n    ix = ix + 1\n    \ny_pred = train_y.copy(deep=True)\ny_pred = y_pred.rename(columns={\"target\": \"prediction\"})\ny_pred[\"prediction\"] = y_oof\nval_score = amex_metric(train_y.target.values, y_pred.prediction.values)\nprint(f\"Amex metric: {val_score}\")\n\nwith open('y_oof_{}.pkl'.format(model_version), 'wb') as outfile:\n    pickle.dump(y_oof, outfile)\n\nwith open('y_test_{}.pkl'.format(model_version), 'wb') as outfile:\n    pickle.dump(y_test, outfile)","metadata":{"execution":{"iopub.status.busy":"2022-07-08T03:22:27.314583Z","iopub.execute_input":"2022-07-08T03:22:27.315307Z","iopub.status.idle":"2022-07-08T03:23:40.785922Z","shell.execute_reply.started":"2022-07-08T03:22:27.31526Z","shell.execute_reply":"2022-07-08T03:23:40.784943Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Submission","metadata":{}},{"cell_type":"code","source":"test['prediction'] = y_test\ntest['prediction'].to_csv('submission_model_{}_val_{}.csv'.format(model_version, val_score), index=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-08T03:23:40.787264Z","iopub.execute_input":"2022-07-08T03:23:40.787774Z","iopub.status.idle":"2022-07-08T03:23:40.856988Z","shell.execute_reply.started":"2022-07-08T03:23:40.787741Z","shell.execute_reply":"2022-07-08T03:23:40.855644Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Confusion Matrix","metadata":{}},{"cell_type":"code","source":"y_oof_binary = (y_oof >= np.percentile(y_oof, 96)).astype(int)  # at top 4%\ny_oof_binary.mean()","metadata":{"execution":{"iopub.status.busy":"2022-07-08T03:23:40.858636Z","iopub.execute_input":"2022-07-08T03:23:40.859002Z","iopub.status.idle":"2022-07-08T03:23:40.868274Z","shell.execute_reply.started":"2022-07-08T03:23:40.858971Z","shell.execute_reply":"2022-07-08T03:23:40.866937Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class_names = [0,1]\ncf = confusion_matrix(train_y, y_oof_binary)\ndisp = ConfusionMatrixDisplay(confusion_matrix=cf, display_labels=class_names)\ndisp.plot()","metadata":{"execution":{"iopub.status.busy":"2022-07-08T03:23:40.869991Z","iopub.execute_input":"2022-07-08T03:23:40.870512Z","iopub.status.idle":"2022-07-08T03:23:41.108984Z","shell.execute_reply.started":"2022-07-08T03:23:40.870476Z","shell.execute_reply":"2022-07-08T03:23:41.108094Z"},"trusted":true},"execution_count":null,"outputs":[]}]}