{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## Objectives\n\nIn order to train this [model](https://www.kaggle.com/code/ragnar123/amex-lgbm-dart-cv-0-7977) on Kaggle Kernels with 16GB RAM instead of 32GB RAM machine, I performed some optimizations to work with these limited resources. This is **final part 3** of the series.\n\n[Part 1](https://www.kaggle.com/code/pham0030/amex-outofmemory-fe-with-lgb-sequence-part1-test) is about test dataset out of memory processing.\n\n[Part 2](https://www.kaggle.com/code/pham0030/amex-outofmemory-fe-with-lgb-sequence-part2-train) is about train dataset out of memory processing with lgb.Sequence.\n\nThank you very much for your work [@ragnar](https://www.kaggle.com/ragnar123).","metadata":{}},{"cell_type":"markdown","source":"## Packages","metadata":{}},{"cell_type":"code","source":"import time\nimport gc\nimport os\nimport joblib\nimport random\n\nimport pandas as pd\nimport numpy as np\nimport h5py\n\nimport lightgbm as lgb","metadata":{"execution":{"iopub.status.busy":"2022-07-14T06:17:08.498419Z","iopub.execute_input":"2022-07-14T06:17:08.498901Z","iopub.status.idle":"2022-07-14T06:17:10.919841Z","shell.execute_reply.started":"2022-07-14T06:17:08.498801Z","shell.execute_reply":"2022-07-14T06:17:10.918628Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Notebook Configuration","metadata":{}},{"cell_type":"code","source":"pd.options.display.max_rows = 2000\npd.options.display.max_seq_items = 2000\n\nDEBUG = False # Change to False for full run","metadata":{"execution":{"iopub.status.busy":"2022-07-14T06:39:00.731957Z","iopub.execute_input":"2022-07-14T06:39:00.733010Z","iopub.status.idle":"2022-07-14T06:39:00.737950Z","shell.execute_reply.started":"2022-07-14T06:39:00.732970Z","shell.execute_reply":"2022-07-14T06:39:00.736946Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Utilities","metadata":{}},{"cell_type":"code","source":"def seed_everything(seed):\n    random.seed(seed)\n    np.random.seed(seed)\n    os.environ['PYTHONHASHSEED'] = str(seed)","metadata":{"execution":{"iopub.status.busy":"2022-07-14T06:39:03.109633Z","iopub.execute_input":"2022-07-14T06:39:03.110236Z","iopub.status.idle":"2022-07-14T06:39:03.115342Z","shell.execute_reply.started":"2022-07-14T06:39:03.110200Z","shell.execute_reply":"2022-07-14T06:39:03.114172Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### LGB Sequence","metadata":{}},{"cell_type":"code","source":"class HDFSequence(lgb.Sequence):\n    def __init__(self, hdf_dataset, batch_size):\n        self.data = hdf_dataset\n        self.batch_size = batch_size\n\n    def __getitem__(self, idx):\n        return self.data[idx]\n\n    def __len__(self):\n        return len(self.data)\n    \ndef create_dataset_from_multiple_hdf_train(input_flist, batch_size):\n    data = []\n    ylist = []\n    for f in input_flist:\n        f = h5py.File(f, 'r')\n        data.append(HDFSequence(f['X'], batch_size))\n        ylist.append(f['Y'][:])\n\n    params = {\n        'bin_construct_sample_cnt': 200000,\n        'max_bin': 255\n    }\n    y = np.concatenate(ylist)\n    dataset = lgb.Dataset(data, label=y, params=params)\n    return dataset\n\ndef create_dataset_from_multiple_hdf_val(input_flist, batch_size, reference):\n    data = []\n    ylist = []\n    for f in input_flist:\n        f = h5py.File(f, 'r')\n        data.append(HDFSequence(f['X'], batch_size))\n        ylist.append(f['Y'][:])\n\n    params = {\n        'bin_construct_sample_cnt': 200000,\n        'max_bin': 255\n    }\n    y = np.concatenate(ylist)\n    dataset = lgb.Dataset(data, label=y, params=params, reference=reference)\n    return dataset","metadata":{"execution":{"iopub.status.busy":"2022-07-14T06:39:04.354144Z","iopub.execute_input":"2022-07-14T06:39:04.354870Z","iopub.status.idle":"2022-07-14T06:39:04.366820Z","shell.execute_reply.started":"2022-07-14T06:39:04.354823Z","shell.execute_reply":"2022-07-14T06:39:04.365932Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### AmEx Metric","metadata":{}},{"cell_type":"code","source":"def amex_metric(y_true, y_pred):\n    labels = np.transpose(np.array([y_true, y_pred]))\n    labels = labels[labels[:, 1].argsort()[::-1]]\n    weights = np.where(labels[:,0]==0, 20, 1)\n    cut_vals = labels[np.cumsum(weights) <= int(0.04 * np.sum(weights))]\n    top_four = np.sum(cut_vals[:,0]) / np.sum(labels[:,0])\n    gini = [0,0]\n    for i in [1,0]:\n        labels = np.transpose(np.array([y_true, y_pred]))\n        labels = labels[labels[:, i].argsort()[::-1]]\n        weight = np.where(labels[:,0]==0, 20, 1)\n        weight_random = np.cumsum(weight / np.sum(weight))\n        total_pos = np.sum(labels[:, 0] *  weight)\n        cum_pos_found = np.cumsum(labels[:, 0] * weight)\n        lorentz = cum_pos_found / total_pos\n        gini[i] = np.sum((lorentz - weight_random) * weight)\n    return 0.5 * (gini[1]/gini[0] + top_four)\n\ndef amex_metric_np(preds, target):\n    indices = np.argsort(preds)[::-1]\n    preds, target = preds[indices], target[indices]\n    weight = 20.0 - target * 19.0\n    cum_norm_weight = (weight / weight.sum()).cumsum()\n    four_pct_mask = cum_norm_weight <= 0.04\n    d = np.sum(target[four_pct_mask]) / np.sum(target)\n    weighted_target = target * weight\n    lorentz = (weighted_target / weighted_target.sum()).cumsum()\n    gini = ((lorentz - cum_norm_weight) * weight).sum()\n    n_pos = np.sum(target)\n    n_neg = target.shape[0] - n_pos\n    gini_max = 10 * n_neg * (n_pos + 20 * n_neg - 19) / (n_pos + 20 * n_neg)\n    g = gini / gini_max\n    return 0.5 * (g + d)\n\ndef lgb_amex_metric(y_pred, y_true):\n    y_true = y_true.get_label()\n    return 'amex_metric', amex_metric(y_true, y_pred), True","metadata":{"execution":{"iopub.status.busy":"2022-07-14T06:39:05.908609Z","iopub.execute_input":"2022-07-14T06:39:05.909000Z","iopub.status.idle":"2022-07-14T06:39:05.928745Z","shell.execute_reply.started":"2022-07-14T06:39:05.908967Z","shell.execute_reply":"2022-07-14T06:39:05.927925Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Custom Callback for 'DART'","metadata":{}},{"cell_type":"markdown","source":"This is custom callback to save best model with 'dart' mode operation in lgbm as it currently not implemented by lgb. Thanks to [@mohammadrahmati](https://www.kaggle.com/mohammadrahmati).","metadata":{}},{"cell_type":"code","source":"global max_score  # not good to use global var, could be improved\nmax_score = 0.75\ndef save_model(fold, log_eval_rounds=100):\n   def callback(env):\n      global max_score\n      iteration = env.iteration\n      score = env.evaluation_result_list[0][2]\n      prev_best_prefix = f'best_data{DATA_VERSION}_lgb{MODEL_VERSION}_fold{fold}_seed{SEED}'\n      if iteration % log_eval_rounds == 0:\n            print(f'Iteration {iteration}, AmEx validation score = {score:.05f}')\n      if score > max_score:\n            max_score = score\n            path = '.'\n            for fname in os.listdir(path):\n                if fname.startswith(prev_best_prefix):\n                    os.remove(os.path.join(path, fname))\n            new_best = prev_best_prefix + f'_iteration{iteration}_score{score:.05f}.pkl'\n            joblib.dump(env.model, new_best)\n   callback.order = 0\n   return callback","metadata":{"execution":{"iopub.status.busy":"2022-07-14T07:03:18.839792Z","iopub.execute_input":"2022-07-14T07:03:18.840780Z","iopub.status.idle":"2022-07-14T07:03:18.851261Z","shell.execute_reply.started":"2022-07-14T07:03:18.840736Z","shell.execute_reply":"2022-07-14T07:03:18.849971Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## LightGBM Model Tuning","metadata":{}},{"cell_type":"markdown","source":"### Data Configuration","metadata":{}},{"cell_type":"code","source":"import re","metadata":{"execution":{"iopub.status.busy":"2022-07-14T07:03:21.858078Z","iopub.execute_input":"2022-07-14T07:03:21.858804Z","iopub.status.idle":"2022-07-14T07:03:21.863009Z","shell.execute_reply.started":"2022-07-14T07:03:21.858767Z","shell.execute_reply":"2022-07-14T07:03:21.862047Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"TRAIN_DATA_PATH = '../input/amex-outofmemory-fe-with-lgb-sequence-part2-train'\n\ntrain_files = []\nfor fname in os.listdir(TRAIN_DATA_PATH):\n    if fname.startswith(f'train_agg') and fname.endswith(f'.h5'):\n        train_files.append(os.path.join(TRAIN_DATA_PATH, fname))\n        \nif len(train_files) == 0:\n    raise(Exception(f'No training data with version: {DATA_VERSION}'))\nelse:  \n    CONSTANTS = re.findall(r'\\d+', os.path.basename(train_files[0]))\n    DATA_VERSION = CONSTANTS[0]\n    SEED = int(CONSTANTS[1])\n    BATCH_SIZE = int(CONSTANTS[2])\n\n    print(f'DATA_VERSION: {DATA_VERSION}')\n    print(f'SEED: {SEED}')\n    seed_everything(SEED)\n    print(f'BATCH_SIZE: {BATCH_SIZE}')\n\ntrain_files","metadata":{"execution":{"iopub.status.busy":"2022-07-14T07:03:23.207241Z","iopub.execute_input":"2022-07-14T07:03:23.208443Z","iopub.status.idle":"2022-07-14T07:03:23.222542Z","shell.execute_reply.started":"2022-07-14T07:03:23.208399Z","shell.execute_reply":"2022-07-14T07:03:23.221619Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Model Configuration","metadata":{}},{"cell_type":"code","source":"MODEL_VERSION=int(time.time())\nprint(f'MODEL_VERSION: {MODEL_VERSION}')\n\nPARAMS = {\n    'objective': 'binary',\n    'metric': 'amex_metric',\n#     'metric': \"binary_logloss\",\n    'boosting': 'dart',  # no early stopping if you use dart\n    'seed': SEED,\n    'num_leaves': 100,\n    'learning_rate': 0.01,\n    'feature_fraction': 0.20,\n    'bagging_freq': 10,\n    'bagging_fraction': 0.50,\n    'n_jobs': -1,\n    'lambda_l2': 2,\n    'min_data_in_leaf': 40,\n#     'device': 'gpu',  # bug\n    }","metadata":{"execution":{"iopub.status.busy":"2022-07-14T07:03:28.336709Z","iopub.execute_input":"2022-07-14T07:03:28.337300Z","iopub.status.idle":"2022-07-14T07:03:28.344578Z","shell.execute_reply.started":"2022-07-14T07:03:28.337267Z","shell.execute_reply":"2022-07-14T07:03:28.343579Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if DEBUG:\n    STOPPING_ROUNDS = 10\n    LOG_EVAL_ROUNDS = 5\n    NUM_BOOST_ROUND = 100\nelse:\n    STOPPING_ROUNDS = 500  # not work in 'dart', implemented with save_model(fold) instead\n    LOG_EVAL_ROUNDS = 100\n    NUM_BOOST_ROUND = 12000","metadata":{"execution":{"iopub.status.busy":"2022-07-14T07:03:29.571249Z","iopub.execute_input":"2022-07-14T07:03:29.571699Z","iopub.status.idle":"2022-07-14T07:03:29.577989Z","shell.execute_reply.started":"2022-07-14T07:03:29.571660Z","shell.execute_reply":"2022-07-14T07:03:29.576734Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Training...","metadata":{}},{"cell_type":"markdown","source":"Here you could spin out 5 notebooks to train **Concurrently** !","metadata":{}},{"cell_type":"code","source":"%%time\n# # hardcoded\n\n# folds_to_train = range(5)  # this take too much time on training on all folds\n\nfolds_to_train = [0]\n# folds_to_train = [1]\n# folds_to_train = [2]\n# folds_to_train = [3]\n# folds_to_train = [4]\n\n\nfor fold in folds_to_train:\n    print(' ')\n    print('-'*50)\n    print('-'*50)\n    print(f'Training fold {fold}...')\n    global max_score  # not good to use global variables\n    max_score = 0.75  # reset global max_score for this fold\n    val_set = [train_files[fold]]\n    train_set = [file for file in train_files if file not in val_set]\n    \n    print('Validation data: ', val_set)\n    print('Training data', train_set)\n    # Create lgb.Sequence\n    lgb_train = create_dataset_from_multiple_hdf_train(train_set, batch_size=BATCH_SIZE)\n    lgb_val = create_dataset_from_multiple_hdf_val(val_set, batch_size=BATCH_SIZE, reference=lgb_train)\n    \n    print('-'*10)\n    evals_result={}\n    model = lgb.train(\n            params = PARAMS,\n            train_set = lgb_train,\n            num_boost_round = NUM_BOOST_ROUND,\n            valid_sets = lgb_val,\n            feval = lgb_amex_metric,\n            callbacks=[\n#                     lgb.early_stopping(stopping_rounds=STOPPING_ROUNDS),  # not working for 'dart'\n                    save_model(fold, log_eval_rounds=LOG_EVAL_ROUNDS),\n#                     lgb.log_evaluation(LOG_EVAL_ROUNDS),\n                    lgb.record_evaluation(evals_result),\n            ]\n           )\n    model_fname = f'last_data{DATA_VERSION}_lgb{MODEL_VERSION}_fold{fold}_seed{SEED}.pkl'\n    joblib.dump(model, model_fname)\n    \n    joblib.dump(evals_result, f'evals_result_data{DATA_VERSION}_lgb{MODEL_VERSION}_fold{fold}_seed{SEED}.pkl')\n    \n    del lgb_train\n    del lgb_val\n    del model\n    gc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-07-14T07:03:31.521954Z","iopub.execute_input":"2022-07-14T07:03:31.522650Z","iopub.status.idle":"2022-07-14T07:06:03.803175Z","shell.execute_reply.started":"2022-07-14T07:03:31.522611Z","shell.execute_reply":"2022-07-14T07:06:03.801822Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"models_path = []\npath='.'\nfor fname in os.listdir(path):\n    if fname.startswith(f'best_data{DATA_VERSION}_lgb{MODEL_VERSION}') and fname.endswith(f'.pkl'):\n        print(fname)\n        models_path.append(fname)","metadata":{"execution":{"iopub.status.busy":"2022-07-14T07:07:03.414809Z","iopub.execute_input":"2022-07-14T07:07:03.415202Z","iopub.status.idle":"2022-07-14T07:07:03.423847Z","shell.execute_reply.started":"2022-07-14T07:07:03.415172Z","shell.execute_reply":"2022-07-14T07:07:03.422373Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"models_path","metadata":{"execution":{"iopub.status.busy":"2022-07-14T07:07:04.414894Z","iopub.execute_input":"2022-07-14T07:07:04.415417Z","iopub.status.idle":"2022-07-14T07:07:04.421876Z","shell.execute_reply.started":"2022-07-14T07:07:04.415369Z","shell.execute_reply":"2022-07-14T07:07:04.420978Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Prediction","metadata":{}},{"cell_type":"markdown","source":"### Test Configuration","metadata":{}},{"cell_type":"code","source":"TEST_DATA_PATH = \"../input/amex-outofmemory-fe-with-lgb-sequence-part1-test/test_fe.csv.gzip\"\nCHUNKSIZE=200000","metadata":{"execution":{"iopub.status.busy":"2022-07-14T07:07:21.400431Z","iopub.execute_input":"2022-07-14T07:07:21.400859Z","iopub.status.idle":"2022-07-14T07:07:21.405670Z","shell.execute_reply.started":"2022-07-14T07:07:21.400823Z","shell.execute_reply":"2022-07-14T07:07:21.404727Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Predicting...","metadata":{}},{"cell_type":"code","source":"%%time\nall_test_predictions = {}\n\nfor idx_fold, model_path in enumerate(models_path):\n    print('')\n    print('-'*50)\n    print('Loading model...', model_path)\n    model = joblib.load(model_path)\n    \n    test_predictions_df = pd.DataFrame({'customer_ID': [], 'prediction': []})\n    print('Predicting...')\n    with pd.read_csv(TEST_DATA_PATH, index_col='customer_ID', chunksize=CHUNKSIZE, compression='gzip') as reader:\n        for test_part_df in reader:\n            print(test_part_df.shape)\n            y_pred = model.predict(test_part_df)\n            test_predictions_part_df = pd.DataFrame({'customer_ID': test_part_df.index, 'prediction': y_pred})\n            test_predictions_df = pd.concat([ test_predictions_df, test_predictions_part_df])\n            \n    fname = model_path[:-3]\n    fname = fname+'csv'\n    fname = os.path.basename(fname)\n    print(f'Saving {fname}...')\n    test_predictions_df.to_csv(fname, index=False)\n    all_test_predictions[idx_fold] = test_predictions_df\n    del test_predictions_df\n    gc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-07-14T07:07:29.012470Z","iopub.execute_input":"2022-07-14T07:07:29.013249Z","iopub.status.idle":"2022-07-14T07:12:22.825856Z","shell.execute_reply.started":"2022-07-14T07:07:29.013196Z","shell.execute_reply":"2022-07-14T07:12:22.824453Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Combine results","metadata":{}},{"cell_type":"code","source":"%%time\nprint('Calculate ensembled predictions...')\nn_folds = len(all_test_predictions)\nensembled_test_predictions = all_test_predictions[0]\nfor i in range(1, n_folds):\n    ensembled_test_predictions['prediction'] += all_test_predictions[i]['prediction']\nensembled_test_predictions['prediction'] = ensembled_test_predictions['prediction'] / n_folds\nensembled_test_predictions.to_csv(f'combined_all_folds_data{DATA_VERSION}_lgb{MODEL_VERSION}_seed{SEED}.csv', index = False)","metadata":{"execution":{"iopub.status.busy":"2022-07-14T07:18:26.433703Z","iopub.execute_input":"2022-07-14T07:18:26.434129Z","iopub.status.idle":"2022-07-14T07:18:31.796875Z","shell.execute_reply.started":"2022-07-14T07:18:26.434098Z","shell.execute_reply":"2022-07-14T07:18:31.795297Z"},"trusted":true},"execution_count":null,"outputs":[]}]}