{"cells":[{"metadata":{},"cell_type":"markdown","source":"### In this notebook, we look at a bug in xgboost (1.2.1 or older) GPU mode. When the input data is `column major`, the performance will be significantly worse. Please change it to `row major` before training.\n\n### Please note that this bug is resolved in the new xgboost 1.3.0 version [#6459](https://github.com/dmlc/xgboost/pull/6459)"},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import xgboost as xgb\nimport cuml\nimport cupy\n\nfrom cuml.metrics import roc_auc_score\nfrom cuml.datasets.classification import make_classification","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"def print_version(*x):\n    for i in x:\n        print(i, eval(f'{i}.__version__'))\n        \ndef print_data_info(*x):\n    \n    for i in x:\n        data = eval(i)\n        major = 'row major' if data.flags.c_contiguous else 'column major'\n        print(i, type(data), data.shape, major)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Let's check out the version of libraries."},{"metadata":{"trusted":true},"cell_type":"code","source":"print_version('xgb', 'cuml', 'cupy')","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Create some synthetic data for binary classification."},{"metadata":{"trusted":true},"cell_type":"code","source":"%%time\n\nX, y = make_classification ( n_classes = 2,\n                             n_features = 10,\n                             n_samples = 10000,\n                             random_state = 0 )\n\nprint_data_info('X', 'y')","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Train a simple xgboost model and return training AUC. "},{"metadata":{"trusted":true},"cell_type":"code","source":"def train_xgb(X, y):\n    params = {'eta': 0.1,\n              'max_depth': 3,\n              'objective': 'binary:logistic',\n              'eval_metric': 'auc',\n              'tree_method': 'gpu_hist',\n             }\n\n    dtrain = xgb.DMatrix(data=X, label=y)\n    bst = xgb.train(params, dtrain=dtrain,\n                    num_boost_round=10)\n\n    score = roc_auc_score(y, bst.predict(dtrain))\n    print(f\"training AUC = {score:.3f}\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_xgb(X, y)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"The performance is quite poor! Let's check out the layout of data and change it."},{"metadata":{"trusted":true},"cell_type":"code","source":"print('Befor change:')\nprint_data_info('X')\n\nprint('After change:')\nX = cupy.ascontiguousarray(X)\nprint_data_info('X')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_xgb(X, y)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Voila! I was astonished when I first found this."}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}