{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84493,"databundleVersionId":9871156,"sourceType":"competition"},{"sourceId":207815053,"sourceType":"kernelVersion"}],"dockerImageVersionId":30786,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Introduction\n\nThis notebook is a proof of concept showing that it is possible to perform **\"extended\" training of XGBoost on a larger data set**. Using standard data reading method (e.g. reading a partition or multiple partitions itno RAM) was expensive and I could barely fit XGboost on a dataset of 2 partitions. Notably, the validation data set needs to be large enough to be \"representative\" but also small enough to be able to run on the Kaggle kernel.\n\n**Strategy**: Use a pytorch-style dataloader, see [XGBoost External Memory](https://xgboost.readthedocs.io/en/stable/tutorials/external_memory.html) and the `Iterator` class below.\nNotes:\n* The `Iterator` class essentially pre-loads / stores the data on the disk. The Kaggle kernel has a limit of 20GB on the `working` folder. We can work around this by creating a temporary directory `'/kaggle/tmp'` where this cache will be stored. It is unknown to me how much can be stored there but I could successfully run this kernel with date_ids 0-1300 as training and date_ids 1450-1650 for validation.\n* This requires the original [`xgboost.train()`](https://xgboost.readthedocs.io/en/stable/python/python_api.html#xgboost.train) implementation and the [`xgboost.Dmatrix`](https://xgboost.readthedocs.io/en/stable/python/python_api.html#xgboost.DMatrix) data format, rather than [`xgboost.XGBRegressor`](https://xgboost.readthedocs.io/en/stable/python/python_api.html#xgboost.XGBRegressor). For prediction in the inference, load the fitted model into `xgboost.XGBRegressor`, I will follow up with an inference notebook soon!\n\n**Other comments**:\n* Kaggle CPU runtime for this: ~1-1.5 hrs. LB score: 0.0047\n* I was not able to use the GPU yet. Maybe the tree parameters are too \"big\". Comments/Suggestions appreciated!\n\n**Pipeline**:\n* Preprocessing: [https://www.kaggle.com/code/kevinlam/2024-jane-street-preprocessing-public](https://www.kaggle.com/code/kevinlam/2024-jane-street-preprocessing-public)\n* Model fitting: [this notebook]\n* Inference: to come.","metadata":{}},{"cell_type":"code","source":"import os, pickle\nimport numpy as np\nimport pandas as pd\nimport polars as pl\n\nimport xgboost as xgb\nfrom typing import List, Callable\n\nkaggle_dir = '/kaggle/input/jane-street-real-time-market-data-forecasting/'\ndata_dir = '/kaggle/input/2024-jane-street-preprocessing-public'\nworking_dir = '/kaggle/working'","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-11-16T17:37:13.774548Z","iopub.execute_input":"2024-11-16T17:37:13.775124Z","iopub.status.idle":"2024-11-16T17:37:15.786851Z","shell.execute_reply.started":"2024-11-16T17:37:13.775072Z","shell.execute_reply":"2024-11-16T17:37:15.785643Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Create a temporary folder where data loader cache will be stored (to avoid the 20GB limit)\ntmp_dir = '/kaggle/tmp'\n\nprint(\"Current Directory: \", os.getcwd())\nos.mkdir(tmp_dir)\nos.chdir(tmp_dir)\nprint(\"New Current Directory: \", os.getcwd())\n\nassert os.path.isdir('/kaggle/tmp')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-16T17:37:15.789131Z","iopub.execute_input":"2024-11-16T17:37:15.789824Z","iopub.status.idle":"2024-11-16T17:37:15.799281Z","shell.execute_reply.started":"2024-11-16T17:37:15.789778Z","shell.execute_reply":"2024-11-16T17:37:15.797441Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Global Parameters","metadata":{}},{"cell_type":"code","source":"# Feature columns\nkey_cols = [\"date_id\", \"time_id\", \"symbol_id\"]\nweight_cols = ['weight']\n\ntarget_cols = [f\"responder_{i}\" for i in range(9)]\nlag_cols = [x + \"_lag_1\" for x in target_cols]\nfeature_cols = [f\"feature_{i:02d}\" for i in range(79)]\n\nother_cols = ['date_id_cos360', 'time_id_cos360', 'date_id_cos180', 'time_id_cos180', 'date_id_cos90', 'time_id_cos90', 'time_id_adj']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-16T17:38:21.230372Z","iopub.execute_input":"2024-11-16T17:38:21.230987Z","iopub.status.idle":"2024-11-16T17:38:21.240401Z","shell.execute_reply.started":"2024-11-16T17:38:21.230934Z","shell.execute_reply":"2024-11-16T17:38:21.238761Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Overall Configuration\nCONFIG = {\n    'seed': 49,\n    'target_col': [\"responder_6\"],\n    'feature_cols': key_cols + feature_cols + lag_cols + other_cols,\n    'device': 'cpu',\n    'fill_null': 'zero', # 'zero' in this code corresponds to leaving nan's as nans. Use 0.0 to actually fill the missing values with 0.0\n    'train_idx': [1000, 1699],\n    'valid_idx': [750, 950]\n}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-16T17:38:22.470252Z","iopub.execute_input":"2024-11-16T17:38:22.470781Z","iopub.status.idle":"2024-11-16T17:38:22.478418Z","shell.execute_reply.started":"2024-11-16T17:38:22.470738Z","shell.execute_reply":"2024-11-16T17:38:22.476748Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"if CONFIG['device'] == 'cuda':\n    grow_policy='depthwise'\n    subsample=0.1\n    colsample_bytree=0.8\nelif CONFIG['device'] == 'cpu':\n    grow_policy = 'depthwise'\n    subsample = 0.8\n    colsample_bytree = 0.8\n\n# XGB parameters\nparams = {\n    'objective': 'reg:squarederror',\n    'learning_rate': 0.05,\n    'max_depth': 6,\n    'tree_method': 'hist', 'grow_policy': grow_policy,\n    'subsample': subsample, 'colsample_bytree': colsample_bytree,\n    'reg_alpha': 0.5, 'reg_lambda': 0.5,\n    'seed': CONFIG['seed'],\n    'device': CONFIG['device']\n}\n\n# parameters for xgb.train()\ntrain_params = {\n    'num_boost_round': 200,\n    'early_stopping_rounds': 100\n}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-16T17:38:23.758212Z","iopub.execute_input":"2024-11-16T17:38:23.758717Z","iopub.status.idle":"2024-11-16T17:38:23.767342Z","shell.execute_reply.started":"2024-11-16T17:38:23.758674Z","shell.execute_reply":"2024-11-16T17:38:23.765907Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def r2_xgb_analysis(y_true, y_pred, sample_weight, \n                    target_col=CONFIG['target_col'].index('responder_6'), multi=len(CONFIG['target_col'])):\n    \"\"\"\n    R2 evaluation for direct predictions. Also for the case when we use multi column predictions (multiple responders)\n    \"\"\"\n    \n    y_true = np.array(y_true).reshape(-1, multi)[:,target_col]\n    y_pred = np.clip(np.array(y_pred).reshape(-1, multi)[:,target_col], -5, 5)\n    sample_weight = np.array(sample_weight)\n    \n    enum = np.sum(sample_weight*((y_true - y_pred)**2))\n    denom = np.sum(sample_weight*(y_true**2))\n    r2 = 1 - enum/denom\n    return -r2\n\ndef r2_xgb_multi(y_pred, dtrain, \n                 target_col=CONFIG['target_col'].index('responder_6'), \n                 multi=len(CONFIG['target_col'])):\n    \"\"\"\n    Custom R^2 evaluation function for XGBoost, for a specific target column in a multi-output setup. For xgboost.train version\n\n    Parameters:\n    - y_pred: Predictions from XGBoost (flattened array).\n    - dtrain: DMatrix object containing true labels and sample weights.\n    - target_col: Index of the target column to evaluate.\n    - multi: Total number of target columns.\n\n    Returns:\n    - A tuple with metric name and the negative R^2 score (since XGBoost minimizes metrics).\n    \"\"\"\n    y_true = dtrain.get_label().reshape(-1, multi)[:, target_col]\n    y_pred = np.clip(y_pred.reshape(-1, multi)[:, target_col], -5, 5)\n    sample_weight = dtrain.get_weight()\n    \n    # Calculate R^2\n    enum = np.sum(sample_weight * ((y_true - y_pred) ** 2))\n    denom = np.sum(sample_weight * (y_true ** 2))\n    r2 = 1 - enum / denom\n\n    # Return metric name and value (negative R^2 for minimization)\n    return 'r2_xgb', -r2\n\ndef get_data_split(data, feature_cols, target_cols, weight_col='weight', symbol_col='symbol_id'):\n    X_data = data.select(pl.col(feature_cols))\n    y_data = data.select(pl.col(target_cols))\n    w_data = data.select(pl.col(weight_col)).get_columns()[0]\n    sy_data = data.select(pl.col(symbol_col))\n\n    return X_data, y_data, w_data, sy_data","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-16T16:49:53.959976Z","iopub.execute_input":"2024-11-16T16:49:53.960351Z","iopub.status.idle":"2024-11-16T16:49:53.971159Z","shell.execute_reply.started":"2024-11-16T16:49:53.960317Z","shell.execute_reply":"2024-11-16T16:49:53.969914Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 0. XGB Data Loader","metadata":{}},{"cell_type":"code","source":"date_ids = pd.concat([pd.read_csv(os.path.join(data_dir, f'dates_{i}.csv')) for i in list(range(10))]).reset_index(drop=True)\n\ntrain_idx = list(date_ids[CONFIG['train_idx'][0]:CONFIG['train_idx'][1]]['date_id'])\nvalid_idx = list(date_ids[CONFIG['valid_idx'][0]:CONFIG['valid_idx'][1]]['date_id'])\n\n# train_idx = list(date_ids[1190:1200]['date_id'])\n# valid_idx = list(date_ids[1590:1600]['date_id'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-16T17:38:26.776167Z","iopub.execute_input":"2024-11-16T17:38:26.77717Z","iopub.status.idle":"2024-11-16T17:38:26.844283Z","shell.execute_reply.started":"2024-11-16T17:38:26.777112Z","shell.execute_reply":"2024-11-16T17:38:26.842999Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class Iterator(xgb.DataIter):\n    def __init__(self, partitions: List[str], feature_cols: List[str], target_col: List[str], weight_col='weight', fill_null=None):\n        self._file_paths = partitions\n        self._it = 0\n        self.feature_cols = feature_cols\n        self.target_col = target_col\n        self.weight_col = weight_col\n        self.fill_null = fill_null\n        super().__init__(cache_prefix=os.path.join(\".\", \"cache\"))\n\n    def next(self, input_data: Callable):\n        # return 0 to let XGBoost know this is the end of iteration\n        if self._it == len(self._file_paths):\n            return 0\n    \n        # load formatted data\n        # train = pl.read_parquet(f'/kaggle/input/2024-jane-street-preprocessing/training/date_id={self._file_paths[self._it]}/00000000.parquet').fill_null(self.fill_null)\n        train = pl.read_parquet(os.path.join(data_dir, f'training/date_id={self._file_paths[self._it]}/00000000.parquet')).fill_null(self.fill_null)\n        \n        X_train = train.select(pl.col(self.feature_cols)).to_pandas()\n        y_train = train.select(pl.col(self.target_col)).to_pandas()\n        w_train = train.select(pl.col(self.weight_col)).get_columns()[0].to_pandas()\n        \n        input_data(data=X_train, label=y_train, weight=w_train)\n        self._it += 1\n        # Return 1 to let XGBoost know we haven't seen all the files yet.\n        return 1\n\n    def reset(self):\n        \"\"\"Reset the iterator to its beginning\"\"\"\n        self._it = 0","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-16T03:37:03.093441Z","iopub.execute_input":"2024-11-16T03:37:03.093945Z","iopub.status.idle":"2024-11-16T03:37:03.104352Z","shell.execute_reply.started":"2024-11-16T03:37:03.093868Z","shell.execute_reply":"2024-11-16T03:37:03.102855Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"Preparing XGB Data Loader...\")\nit = Iterator(train_idx, feature_cols=CONFIG['feature_cols'], target_col=CONFIG['target_col'], fill_null=CONFIG['fill_null'])\ndtrain = xgb.DMatrix(it)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-16T03:38:19.714027Z","iopub.execute_input":"2024-11-16T03:38:19.714471Z","iopub.status.idle":"2024-11-16T03:38:20.983324Z","shell.execute_reply.started":"2024-11-16T03:38:19.714431Z","shell.execute_reply":"2024-11-16T03:38:20.982269Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 1. Data Setup","metadata":{}},{"cell_type":"code","source":"print(\"Preparing Validation Data ...\")\nvalid = pl.concat(\n    [pl.read_parquet(os.path.join(data_dir, f'training/date_id={i}/00000000.parquet')).fill_null(CONFIG['fill_null']) for i in valid_idx]\n)\n\nX_valid, y_valid, w_valid, sy_valid = get_data_split(valid, CONFIG['feature_cols'], CONFIG['target_col'])\n\ndvalid = xgb.DMatrix(data=X_valid.to_pandas(), label=y_valid.to_pandas(), weight=w_valid.to_pandas())\nwatchlist = [(dvalid, 'eval')]\n\ndel valid, X_valid, sy_valid, y_valid, w_valid, dvalid","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-16T03:38:23.852879Z","iopub.execute_input":"2024-11-16T03:38:23.853313Z","iopub.status.idle":"2024-11-16T03:38:25.287278Z","shell.execute_reply.started":"2024-11-16T03:38:23.853274Z","shell.execute_reply":"2024-11-16T03:38:25.285961Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 2. Training","metadata":{}},{"cell_type":"code","source":"print(\"Training The Model...\")\n\nmodel = xgb.train(\n    params=params,\n    dtrain=dtrain,\n    evals=watchlist,\n    verbose_eval=5,\n    custom_metric=r2_xgb_multi,\n    **train_params\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-16T03:35:12.849103Z","iopub.execute_input":"2024-11-16T03:35:12.849571Z","iopub.status.idle":"2024-11-16T03:35:12.855465Z","shell.execute_reply.started":"2024-11-16T03:35:12.849531Z","shell.execute_reply":"2024-11-16T03:35:12.853982Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 3. Result","metadata":{}},{"cell_type":"code","source":" # Display the best iteration and corresponding score\nprint(f\"Best iteration: {model.best_iteration}\")\nprint(f\"Best validation score: {-model.best_score}\")\n\nprint(\"Saving Model...\")\nresult = {\n    \"model\": model,\n    \"params\": params,\n    \"train_params\": train_params,\n    \"CONFIG\": CONFIG\n}\nwith open(os.path.join(working_dir, f\"model_{CONFIG['train_idx'][0]}_{CONFIG['train_idx'][1]}vs{CONFIG['valid_idx'][0]}_{CONFIG['valid_idx'][1]}.pkl\"), \"wb\") as fp:\n    pickle.dump(result, fp)\n\nmodel.save_model(os.path.join(working_dir, f\"model_{CONFIG['train_idx'][0]}_{CONFIG['train_idx'][1]}vs{CONFIG['valid_idx'][0]}_{CONFIG['valid_idx'][1]}.json\"))","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 4. Playground","metadata":{}},{"cell_type":"code","source":"# # If you want to use manually selected features\n# features = pd.read_csv(os.path.join(kaggle_dir, 'features.csv'))\n# responders = pd.read_csv(os.path.join(kaggle_dir, 'responders.csv'))\n\n# # for i in range(17):\n# #     print(i, len(features.query(f'tag_{i} == True')))\n\n# feature_cols_manual = list(features.query('tag_0 == True or tag_2 == True or tag_4 == True').reset_index(drop=True)['feature'])","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}