{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## Objectives\n\nIn order to train this [model](https://www.kaggle.com/code/ragnar123/amex-lgbm-dart-cv-0-7977) on Kaggle Kernels with 16GB RAM instead of 32GB RAM machine, I performed some optimizations to work with these limited resources. This is part 2 of the series.\n\nThank you very much for your work [@ragnar](https://www.kaggle.com/ragnar123).","metadata":{}},{"cell_type":"markdown","source":"## Part 2: Train Data Processing","metadata":{}},{"cell_type":"markdown","source":"Since we could work with train dataset on 16GB memory, we don't have to perform aggregration by chunk of colums like test dataset. Part 2 could run independently of [part 1](https://www.kaggle.com/code/pham0030/amex-outofmemory-fe-with-lgb-sequence-part1-test/notebook). Be aware of the order of joining features though.","metadata":{}},{"cell_type":"markdown","source":"### Configuration","metadata":{}},{"cell_type":"code","source":"import os\nimport gc\nimport pandas as pd\nimport numpy as np\nfrom tqdm.auto import tqdm\n# add progress_apply method to DataFrame objects\ntqdm.pandas()\n\n## hardcoded features to reduce memory read\nfeatures = [\"customer_ID\", \"S_2\", \"P_2\", \"D_39\", \"B_1\", \"B_2\", \"R_1\", \"S_3\", \"D_41\", \"B_3\", \"D_42\", \"D_43\", \n            \"D_44\", \"B_4\", \"D_45\", \"B_5\", \"R_2\", \"D_46\", \"D_47\", \"D_48\", \"D_49\", \"B_6\", \n            \"B_7\", \"B_8\", \"D_50\", \"D_51\", \"B_9\", \"R_3\", \"D_52\", \"P_3\", \"B_10\", \"D_53\", \n            \"S_5\", \"B_11\", \"S_6\", \"D_54\", \"R_4\", \"S_7\", \"B_12\", \"S_8\", \"D_55\", \"D_56\", \n            \"B_13\", \"R_5\", \"D_58\", \"S_9\", \"B_14\", \"D_59\", \"D_60\", \"D_61\", \"B_15\", \"S_11\", \n            \"D_62\", \"D_63\", \"D_64\", \"D_65\", \"B_16\", \"B_17\", \"B_18\", \"B_19\", \"D_66\", \"B_20\", \n            \"D_68\", \"S_12\", \"R_6\", \"S_13\", \"B_21\", \"D_69\", \"B_22\", \"D_70\", \"D_71\", \"D_72\", \n            \"S_15\", \"B_23\", \"D_73\", \"P_4\", \"D_74\", \"D_75\", \"D_76\", \"B_24\", \"R_7\", \"D_77\", \n            \"B_25\", \"B_26\", \"D_78\", \"D_79\", \"R_8\", \"R_9\", \"S_16\", \"D_80\", \"R_10\", \"R_11\", \n            \"B_27\", \"D_81\", \"D_82\", \"S_17\", \"R_12\", \"B_28\", \"R_13\", \"D_83\", \"R_14\", \"R_15\", \n            \"D_84\", \"R_16\", \"B_29\", \"B_30\", \"S_18\", \"D_86\", \"D_87\", \"R_17\", \"R_18\", \"D_88\", \n            \"B_31\", \"S_19\", \"R_19\", \"B_32\", \"S_20\", \"R_20\", \"R_21\", \"B_33\", \"D_89\", \"R_22\", \n            \"R_23\", \"D_91\", \"D_92\", \"D_93\", \"D_94\", \"R_24\", \"R_25\", \"D_96\", \"S_22\", \"S_23\", \n            \"S_24\", \"S_25\", \"S_26\", \"D_102\", \"D_103\", \"D_104\", \"D_105\", \"D_106\", \"D_107\", \n            \"B_36\", \"B_37\", \"R_26\", \"R_27\", \"B_38\", \"D_108\", \"D_109\", \"D_110\", \"D_111\", \n            \"B_39\", \"D_112\", \"B_40\", \"S_27\", \"D_113\", \"D_114\", \"D_115\", \"D_116\", \"D_117\", \n            \"D_118\", \"D_119\", \"D_120\", \"D_121\", \"D_122\", \"D_123\", \"D_124\", \"D_125\", \"D_126\", \n            \"D_127\", \"D_128\", \"D_129\", \"B_41\", \"B_42\", \"D_130\", \"D_131\", \"D_132\", \"D_133\", \n            \"R_28\", \"D_134\", \"D_135\", \"D_136\", \"D_137\", \"D_138\", \"D_139\", \"D_140\", \"D_141\", \n            \"D_142\", \"D_143\", \"D_144\", \"D_145\"]\n\nfeatures.remove('customer_ID')\nfeatures.remove('S_2')\ncat_features = [\n    \"B_30\",\n    \"B_38\",\n    \"D_114\",\n    \"D_116\",\n    \"D_117\",\n    \"D_120\",\n    \"D_126\",\n    \"D_63\",\n    \"D_64\",\n    \"D_66\",\n    \"D_68\",\n]\nnum_features = [col for col in features if col not in cat_features]\nprint(f'Number of original features (excludes customer_ID, S_2): {len(features)}')\nprint(f'Number of categorical features: {len(cat_features)}')\nprint(f'Number of numeric features: {len(num_features)}')","metadata":{"execution":{"iopub.status.busy":"2022-07-13T03:01:26.767638Z","iopub.execute_input":"2022-07-13T03:01:26.768434Z","iopub.status.idle":"2022-07-13T03:01:26.925623Z","shell.execute_reply.started":"2022-07-13T03:01:26.768386Z","shell.execute_reply":"2022-07-13T03:01:26.924552Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nprint('Starting training feature engineer...')\ntrain = pd.read_parquet('../input/amex-data-integer-dtypes-parquet-format/train.parquet')\n\ntrain_num_agg = train.groupby(\"customer_ID\")[num_features].agg(['mean', 'std', 'min', 'max', 'last'])\ntrain_num_agg.columns = ['_'.join(x) for x in train_num_agg.columns]\n# train_num_agg.reset_index(inplace = True)\n\ntrain_cat_agg = train.groupby(\"customer_ID\")[cat_features].agg(['count', 'last', 'nunique'])\ntrain_cat_agg.columns = ['_'.join(x) for x in train_cat_agg.columns]\n# train_cat_agg.reset_index(inplace = True)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T15:18:46.663848Z","iopub.execute_input":"2022-07-12T15:18:46.664842Z","iopub.status.idle":"2022-07-12T15:20:16.412358Z","shell.execute_reply.started":"2022-07-12T15:18:46.664778Z","shell.execute_reply":"2022-07-12T15:20:16.411398Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nprint('Transform cols with float64 to float32 to reduce memory usage...')\ncols = list(train_num_agg.dtypes[train_num_agg.dtypes == 'float64'].index)\ntrain_num_agg.loc[:,cols] = train_num_agg.loc[:,cols].progress_apply(lambda x: x.astype(np.float32))\n\nprint('Transform cols with int64 to int32 to reduce memory usage...')\ncols = list(train_cat_agg.dtypes[train_cat_agg.dtypes == 'int64'].index)\ntrain_cat_agg.loc[:,cols] = train_cat_agg.loc[:,cols].progress_apply(lambda x: x.astype(np.int32))\n\n# Difference features (diff1)\nprint('Get the difference features...')\ntrain_diff = train.loc[:,num_features+['customer_ID']].groupby(['customer_ID']).progress_apply(lambda x: np.diff(x.values[-2:,:], axis = 0).squeeze().astype(np.float32))\nindex = train_diff.index\ncols = [col + '_diff1' for col in train[num_features].columns]\ntrain_diff = pd.DataFrame(train_diff.values.tolist(), columns=cols)\ntrain_diff['customer_ID'] = index","metadata":{"execution":{"iopub.status.busy":"2022-07-12T15:20:16.413946Z","iopub.execute_input":"2022-07-12T15:20:16.414782Z","iopub.status.idle":"2022-07-12T15:22:34.586711Z","shell.execute_reply.started":"2022-07-12T15:20:16.414743Z","shell.execute_reply":"2022-07-12T15:22:34.585752Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Thanks to [@patrico49](https://www.kaggle.com/patrico49) for some nice optimizations","metadata":{}},{"cell_type":"markdown","source":"Merge the data, beware the order so that it would match the test set","metadata":{}},{"cell_type":"code","source":"%%time\nprint('Load label data...')\ntrain_labels = pd.read_csv('../input/amex-default-prediction/train_labels.csv')\n\nprint('Merge train data...')\ntrain = train_num_agg \\\n    .merge(train_diff, how = 'inner', on = 'customer_ID') \\\n    .merge(train_cat_agg, how = 'inner', on = 'customer_ID') \\\n    .merge(train_labels, how = 'inner', on = 'customer_ID')\ntrain","metadata":{"execution":{"iopub.status.busy":"2022-07-12T15:22:34.587949Z","iopub.execute_input":"2022-07-12T15:22:34.588733Z","iopub.status.idle":"2022-07-12T15:23:25.867779Z","shell.execute_reply.started":"2022-07-12T15:22:34.588697Z","shell.execute_reply":"2022-07-12T15:23:25.866258Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del train_num_agg \ndel train_diff\ndel train_cat_agg\ndel train_labels \ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-07-12T15:23:26.167298Z","iopub.execute_input":"2022-07-12T15:23:26.167791Z","iopub.status.idle":"2022-07-12T15:23:26.344448Z","shell.execute_reply.started":"2022-07-12T15:23:26.167749Z","shell.execute_reply":"2022-07-12T15:23:26.343388Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Save the results","metadata":{}},{"cell_type":"code","source":"%%time\ntrain.to_parquet('train_fe.parquet')","metadata":{"execution":{"iopub.status.busy":"2022-07-12T15:23:26.345783Z","iopub.execute_input":"2022-07-12T15:23:26.346315Z","iopub.status.idle":"2022-07-12T15:23:53.374079Z","shell.execute_reply.started":"2022-07-12T15:23:26.346282Z","shell.execute_reply":"2022-07-12T15:23:53.373000Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### LGB Sequence Data Processing","metadata":{}},{"cell_type":"markdown","source":"Though we managed to process the data on memory, training the data with single dataframe will still result in overflow error.\n\n`lgb.Sequence` will help you to overcome this problem by splitting the training dataset. Refer to [this example](https://github.com/microsoft/LightGBM/blob/master/examples/python-guide/dataset_from_multi_hdf5.py) and [lgb.Sequence docs](https://lightgbm.readthedocs.io/en/latest/pythonapi/lightgbm.Sequence.html) for more information.","metadata":{}},{"cell_type":"code","source":"import time\nimport joblib\nimport h5py\nimport random","metadata":{"execution":{"iopub.status.busy":"2022-07-13T03:01:22.966684Z","iopub.execute_input":"2022-07-13T03:01:22.967195Z","iopub.status.idle":"2022-07-13T03:01:23.776817Z","shell.execute_reply.started":"2022-07-13T03:01:22.967152Z","shell.execute_reply":"2022-07-13T03:01:23.775324Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Configuration","metadata":{}},{"cell_type":"code","source":"pd.options.display.max_rows = 2000\npd.options.display.max_seq_items = 2000\nDEBUG = False\nNROWS_DEBUG = 5000\n\nDATA_VERSION=int(time.time())\nprint('Data version: {}'.format(DATA_VERSION))","metadata":{"execution":{"iopub.status.busy":"2022-07-12T15:24:26.469350Z","iopub.execute_input":"2022-07-12T15:24:26.470449Z","iopub.status.idle":"2022-07-12T15:24:26.476483Z","shell.execute_reply.started":"2022-07-12T15:24:26.470395Z","shell.execute_reply":"2022-07-12T15:24:26.475579Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Reload the data if neccessary","metadata":{}},{"cell_type":"code","source":"train = pd.read_parquet('./train_fe.parquet')","metadata":{"execution":{"iopub.status.busy":"2022-07-12T15:24:30.662838Z","iopub.execute_input":"2022-07-12T15:24:30.663511Z","iopub.status.idle":"2022-07-12T15:24:34.528488Z","shell.execute_reply.started":"2022-07-12T15:24:30.663470Z","shell.execute_reply":"2022-07-12T15:24:34.527507Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nif DEBUG:\n    train = train.head(NROWS_DEBUG)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Reconfig features","metadata":{}},{"cell_type":"code","source":"features = [col for col in train.columns if col not in ['customer_ID', 'target']]","metadata":{"execution":{"iopub.status.busy":"2022-07-12T15:24:49.740792Z","iopub.execute_input":"2022-07-12T15:24:49.741965Z","iopub.status.idle":"2022-07-12T15:24:49.747356Z","shell.execute_reply.started":"2022-07-12T15:24:49.741916Z","shell.execute_reply":"2022-07-12T15:24:49.745933Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Important lgb.Sequence Configuration","metadata":{}},{"cell_type":"markdown","source":"**Note:**\n* `BATCH_SIZE` is important number, we should keep track of this as the input parameters for model training. We split the training data into `N_FOLDS` = 5 sequence hdf5 files, since the training data is split prior, we could run training cv-fold parallel to speed up the training later by spinning up to 5 kernels or up to Kaggle limits !.\n\n* `SEED` to introduce and control randomness to our training split","metadata":{}},{"cell_type":"code","source":"BATCH_SIZE=64\nN_FOLDS=5\nSEED = 99","metadata":{"execution":{"iopub.status.busy":"2022-07-13T03:01:09.829027Z","iopub.execute_input":"2022-07-13T03:01:09.829445Z","iopub.status.idle":"2022-07-13T03:01:09.834155Z","shell.execute_reply.started":"2022-07-13T03:01:09.829419Z","shell.execute_reply":"2022-07-13T03:01:09.833241Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Utility","metadata":{}},{"cell_type":"code","source":"def save2hdf(input_data, fname, batch_size=64):\n    \"\"\"Store numpy array to HDF5 file.\n    Please note chunk size settings in the implementation for I/O performance optimization.\n    \"\"\"\n    with h5py.File(fname, 'w') as f:\n        for name, data in input_data.items():\n            nrow, ncol = data.shape\n            if ncol == 1:\n                # Y has a single column and we read it in single shot. So store it as an 1-d array.\n                chunk = (nrow,)\n                data = data.values.flatten()\n            else:\n                # We use random access for data sampling when creating LightGBM Dataset from Sequence.\n                # When accessing any element in a HDF5 chunk, it's read entirely.\n                # To save I/O for sampling, we should keep number of total chunks much larger than sample count.\n                # Here we are just creating a chunk size that matches with batch_size.\n                #\n                # Also note that the data is stored in row major order to avoid extra copy when passing to\n                # lightgbm Dataset.\n                chunk = (batch_size, ncol)\n            f.create_dataset(name, data=data, chunks=chunk, compression='lzf')","metadata":{"execution":{"iopub.status.busy":"2022-07-12T15:25:01.839534Z","iopub.execute_input":"2022-07-12T15:25:01.840175Z","iopub.status.idle":"2022-07-12T15:25:01.847506Z","shell.execute_reply.started":"2022-07-12T15:25:01.840139Z","shell.execute_reply":"2022-07-12T15:25:01.846255Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def seed_everything(seed):\n    random.seed(seed)\n    np.random.seed(seed)\n    os.environ['PYTHONHASHSEED'] = str(seed)","metadata":{"execution":{"iopub.status.busy":"2022-07-13T03:01:04.693950Z","iopub.execute_input":"2022-07-13T03:01:04.694344Z","iopub.status.idle":"2022-07-13T03:01:04.700321Z","shell.execute_reply.started":"2022-07-13T03:01:04.694314Z","shell.execute_reply":"2022-07-13T03:01:04.699361Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Split traing data","metadata":{}},{"cell_type":"code","source":"print('Saving train data as hdf5 files for lgb.Sequence...')","metadata":{"execution":{"iopub.status.busy":"2022-07-13T03:00:30.631568Z","iopub.execute_input":"2022-07-13T03:00:30.632050Z","iopub.status.idle":"2022-07-13T03:00:30.664566Z","shell.execute_reply.started":"2022-07-13T03:00:30.631937Z","shell.execute_reply":"2022-07-13T03:00:30.663855Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"seed_everything(SEED)","metadata":{"execution":{"iopub.status.busy":"2022-07-13T03:01:34.237623Z","iopub.execute_input":"2022-07-13T03:01:34.238054Z","iopub.status.idle":"2022-07-13T03:01:34.243344Z","shell.execute_reply.started":"2022-07-13T03:01:34.238022Z","shell.execute_reply":"2022-07-13T03:01:34.241741Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nfrom sklearn.model_selection import train_test_split\ntrain_files = []\n\nremain_df = train\nfor idx in reversed(range(1, N_FOLDS)):\n    test_size = 1/(idx+1)\n    remain_df, df = train_test_split(remain_df, test_size=test_size, random_state=SEED)\n    fname = f'train_agg_{DATA_VERSION}_seed{SEED}_batchsize{BATCH_SIZE}_{idx}_{N_FOLDS}.h5'\n    print('Saving...', fname)\n    print(df.shape)\n    save2hdf({'Y': df[['target']], 'X': df[features]}, fname, batch_size=BATCH_SIZE)\n    train_files.append(fname)\n    \n# last part\nidx = 0\nfname = f'train_agg_{DATA_VERSION}_seed{SEED}_batchsize{BATCH_SIZE}_{idx}_{N_FOLDS}.h5'\nprint('Saving...', fname)\nprint(remain_df.shape) \nsave2hdf({'Y': remain_df[['target']], 'X': remain_df[features]}, fname, batch_size=BATCH_SIZE)\ntrain_files.append(fname)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T15:25:04.465727Z","iopub.execute_input":"2022-07-12T15:25:04.466475Z","iopub.status.idle":"2022-07-12T15:25:32.558335Z","shell.execute_reply.started":"2022-07-12T15:25:04.466404Z","shell.execute_reply":"2022-07-12T15:25:32.556922Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del train\ndel remain_df\ndel df\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-07-12T15:25:32.560082Z","iopub.execute_input":"2022-07-12T15:25:32.560501Z","iopub.status.idle":"2022-07-12T15:25:32.746659Z","shell.execute_reply.started":"2022-07-12T15:25:32.560463Z","shell.execute_reply":"2022-07-12T15:25:32.745103Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The data is split into 5 folds and ready to be used with `lgb.Sequence` for out of memory training.","metadata":{}},{"cell_type":"markdown","source":"### Sanity Check","metadata":{}},{"cell_type":"markdown","source":"Let's employ a quick sanity check via training with `lgb.Sequence` to verify the out of memory training capability.","metadata":{}},{"cell_type":"code","source":"import lightgbm as lgb","metadata":{"execution":{"iopub.status.busy":"2022-07-12T15:26:01.257946Z","iopub.execute_input":"2022-07-12T15:26:01.258839Z","iopub.status.idle":"2022-07-12T15:26:02.478451Z","shell.execute_reply.started":"2022-07-12T15:26:01.258780Z","shell.execute_reply":"2022-07-12T15:26:02.477006Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class HDFSequence(lgb.Sequence):\n    def __init__(self, hdf_dataset, batch_size):\n        self.data = hdf_dataset\n        self.batch_size = batch_size\n\n    def __getitem__(self, idx):\n        return self.data[idx]\n\n    def __len__(self):\n        return len(self.data)\n    \ndef create_dataset_from_multiple_hdf_train(input_flist, batch_size):\n    data = []\n    ylist = []\n    for f in input_flist:\n        f = h5py.File(f, 'r')\n        data.append(HDFSequence(f['X'], batch_size))\n        ylist.append(f['Y'][:])\n\n    params = {\n        'bin_construct_sample_cnt': 200000,\n        'max_bin': 255\n    }\n    y = np.concatenate(ylist)\n    dataset = lgb.Dataset(data, label=y, params=params)\n    return dataset\n\ndef create_dataset_from_multiple_hdf_val(input_flist, batch_size, reference):\n    data = []\n    ylist = []\n    for f in input_flist:\n        f = h5py.File(f, 'r')\n        data.append(HDFSequence(f['X'], batch_size))\n        ylist.append(f['Y'][:])\n\n    params = {\n        'bin_construct_sample_cnt': 200000,\n        'max_bin': 255\n    }\n    y = np.concatenate(ylist)\n    dataset = lgb.Dataset(data, label=y, params=params, reference=reference)\n    return dataset","metadata":{"execution":{"iopub.status.busy":"2022-07-12T15:26:02.480383Z","iopub.execute_input":"2022-07-12T15:26:02.481471Z","iopub.status.idle":"2022-07-12T15:26:02.493273Z","shell.execute_reply.started":"2022-07-12T15:26:02.481411Z","shell.execute_reply":"2022-07-12T15:26:02.491627Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The validation set","metadata":{}},{"cell_type":"code","source":"val_files = [train_files[0]]\nval_files","metadata":{"execution":{"iopub.status.busy":"2022-07-12T15:26:03.867474Z","iopub.execute_input":"2022-07-12T15:26:03.869026Z","iopub.status.idle":"2022-07-12T15:26:03.878926Z","shell.execute_reply.started":"2022-07-12T15:26:03.868960Z","shell.execute_reply":"2022-07-12T15:26:03.877531Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The training set","metadata":{}},{"cell_type":"code","source":"train_files = [file for file in train_files if file not in val_files ]\ntrain_files","metadata":{"execution":{"iopub.status.busy":"2022-07-12T15:26:05.321645Z","iopub.execute_input":"2022-07-12T15:26:05.322136Z","iopub.status.idle":"2022-07-12T15:26:05.332189Z","shell.execute_reply.started":"2022-07-12T15:26:05.322090Z","shell.execute_reply":"2022-07-12T15:26:05.330389Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"lgb_train = create_dataset_from_multiple_hdf_train(train_files, batch_size=BATCH_SIZE)\nlgb_val = create_dataset_from_multiple_hdf_val(val_files, batch_size=BATCH_SIZE, reference=lgb_train)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T15:26:08.038063Z","iopub.execute_input":"2022-07-12T15:26:08.039552Z","iopub.status.idle":"2022-07-12T15:26:08.063372Z","shell.execute_reply.started":"2022-07-12T15:26:08.039507Z","shell.execute_reply":"2022-07-12T15:26:08.061766Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**`BATCH_SIZE` is very important, should be matched within the pipeline for optimized performance !**!","metadata":{}},{"cell_type":"markdown","source":"### AmEx custom metric","metadata":{}},{"cell_type":"code","source":"def amex_metric(y_true, y_pred):\n    labels = np.transpose(np.array([y_true, y_pred]))\n    labels = labels[labels[:, 1].argsort()[::-1]]\n    weights = np.where(labels[:,0]==0, 20, 1)\n    cut_vals = labels[np.cumsum(weights) <= int(0.04 * np.sum(weights))]\n    top_four = np.sum(cut_vals[:,0]) / np.sum(labels[:,0])\n    gini = [0,0]\n    for i in [1,0]:\n        labels = np.transpose(np.array([y_true, y_pred]))\n        labels = labels[labels[:, i].argsort()[::-1]]\n        weight = np.where(labels[:,0]==0, 20, 1)\n        weight_random = np.cumsum(weight / np.sum(weight))\n        total_pos = np.sum(labels[:, 0] *  weight)\n        cum_pos_found = np.cumsum(labels[:, 0] * weight)\n        lorentz = cum_pos_found / total_pos\n        gini[i] = np.sum((lorentz - weight_random) * weight)\n    return 0.5 * (gini[1]/gini[0] + top_four)\n\ndef amex_metric_np(preds, target):\n    indices = np.argsort(preds)[::-1]\n    preds, target = preds[indices], target[indices]\n    weight = 20.0 - target * 19.0\n    cum_norm_weight = (weight / weight.sum()).cumsum()\n    four_pct_mask = cum_norm_weight <= 0.04\n    d = np.sum(target[four_pct_mask]) / np.sum(target)\n    weighted_target = target * weight\n    lorentz = (weighted_target / weighted_target.sum()).cumsum()\n    gini = ((lorentz - cum_norm_weight) * weight).sum()\n    n_pos = np.sum(target)\n    n_neg = target.shape[0] - n_pos\n    gini_max = 10 * n_neg * (n_pos + 20 * n_neg - 19) / (n_pos + 20 * n_neg)\n    g = gini / gini_max\n    return 0.5 * (g + d)\n\ndef lgb_amex_metric(y_pred, y_true):\n    y_true = y_true.get_label()\n    return 'amex_metric', amex_metric(y_true, y_pred), True","metadata":{"execution":{"iopub.status.busy":"2022-07-12T15:26:09.508237Z","iopub.execute_input":"2022-07-12T15:26:09.508734Z","iopub.status.idle":"2022-07-12T15:26:09.522477Z","shell.execute_reply.started":"2022-07-12T15:26:09.508678Z","shell.execute_reply":"2022-07-12T15:26:09.520859Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Model configuration","metadata":{}},{"cell_type":"code","source":"PARAMS = {\n    'objective': 'binary',\n    'metric': 'amex_metrix',\n#     'boosting': 'dart',\n    'seed': SEED,\n    'num_leaves': 100,\n    'learning_rate': 0.01,\n    'feature_fraction': 0.20,\n    'bagging_freq': 10,\n    'bagging_fraction': 0.50,\n    'n_jobs': -1,\n    'lambda_l2': 2,\n    'min_data_in_leaf': 40\n    }","metadata":{"execution":{"iopub.status.busy":"2022-07-12T15:26:10.424193Z","iopub.execute_input":"2022-07-12T15:26:10.424992Z","iopub.status.idle":"2022-07-12T15:26:10.432313Z","shell.execute_reply.started":"2022-07-12T15:26:10.424935Z","shell.execute_reply":"2022-07-12T15:26:10.430874Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"STOPPING_ROUNDS = 20\nLOG_EVAL_ROUNDS = 10\nNUM_BOOST_ROUND = 100","metadata":{"execution":{"iopub.status.busy":"2022-07-12T15:26:11.591369Z","iopub.execute_input":"2022-07-12T15:26:11.591848Z","iopub.status.idle":"2022-07-12T15:26:11.598211Z","shell.execute_reply.started":"2022-07-12T15:26:11.591810Z","shell.execute_reply":"2022-07-12T15:26:11.596888Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Start sanity-check training lightgbm with lgb.Sequence...')","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nevals_result={}\nmodel = lgb.train(\n            params = PARAMS,\n            train_set = lgb_train,\n            num_boost_round = NUM_BOOST_ROUND,\n            valid_sets = lgb_val,\n            feval = lgb_amex_metric,\n            callbacks=[\n                    lgb.early_stopping(stopping_rounds=STOPPING_ROUNDS),\n                    lgb.log_evaluation(LOG_EVAL_ROUNDS),\n                    lgb.record_evaluation(evals_result)\n            ]\n           )\n\njoblib.dump(model, 'lgbm_{}.pkl'.format(DATA_VERSION))","metadata":{"execution":{"iopub.status.busy":"2022-07-13T05:42:03.225516Z","iopub.execute_input":"2022-07-13T05:42:03.226036Z","iopub.status.idle":"2022-07-13T05:42:03.323971Z","shell.execute_reply.started":"2022-07-13T05:42:03.225920Z","shell.execute_reply":"2022-07-13T05:42:03.322850Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Sanity check training is fine, let's move on to the final [part 3](https://www.kaggle.com/pham0030/amex-outofmemory-fe-with-lgb-sequence-part3-final) of cv-training and predict!","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}