{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84493,"databundleVersionId":9871156,"sourceType":"competition"},{"sourceId":145414,"sourceType":"modelInstanceVersion","modelInstanceId":123281,"modelId":146350},{"sourceId":149880,"sourceType":"modelInstanceVersion","isSourceIdPinned":true,"modelInstanceId":127242,"modelId":150185}],"dockerImageVersionId":30786,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"### Summary for XGBoost Version\n\nThis notebook presents an updated version of my previous LightGBM experiment, utilizing XGBoost to further explore the capabilities of different models for this competition.\n\n**1. Experiment Overview**  \nBuilding upon the LightGBM-based version, I have implemented XGBoost to compare its performance in similar settings. The key differences include changes in hyperparameters and the use of XGBoost’s GPU-accelerated training.\n\n**2. Previous LightGBM Version Summary**  \nThe previous version of the notebook focused on optimizing data loading and implementing K-fold cross-validation to ensure robust model evaluation. The complete version of that notebook is available [LGB version](https://www.kaggle.com/code/dasbro/janestreet-lgbm-dataload-kfold-baseline/notebook).\n\n**3. Future Directions**  \nMoving forward, I plan to explore ensemble methods by combining the results from different models, including LightGBM, XGBoost, and potentially others, to achieve a more robust and generalized prediction.\nSummary for XGBoost Version\nThis notebook presents an updated version of my previous LightGBM experiment, utilizing XGBoost to further explore the capabilities of different models for this competition.","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport polars as pl\nimport pandas as pd\nimport lightgbm as lgb\nimport xgboost as xgb\nimport os\n\nimport kaggle_evaluation.jane_street_inference_server\n\nimport joblib","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-10-29T06:19:21.044991Z","iopub.execute_input":"2024-10-29T06:19:21.045485Z","iopub.status.idle":"2024-10-29T06:19:25.906419Z","shell.execute_reply.started":"2024-10-29T06:19:21.045440Z","shell.execute_reply":"2024-10-29T06:19:25.904991Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"xgb.__version__","metadata":{"execution":{"iopub.status.busy":"2024-10-29T06:19:25.908709Z","iopub.execute_input":"2024-10-29T06:19:25.909301Z","iopub.status.idle":"2024-10-29T06:19:25.921735Z","shell.execute_reply.started":"2024-10-29T06:19:25.909255Z","shell.execute_reply":"2024-10-29T06:19:25.920566Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"TARGET = 'responder_6'\nFEAT_COLS = [f\"feature_{i:02d}\" for i in range(79)]","metadata":{"execution":{"iopub.status.busy":"2024-10-29T06:19:25.923393Z","iopub.execute_input":"2024-10-29T06:19:25.923880Z","iopub.status.idle":"2024-10-29T06:19:25.933890Z","shell.execute_reply.started":"2024-10-29T06:19:25.923823Z","shell.execute_reply":"2024-10-29T06:19:25.932627Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def load_data(date_id_range=None, time_id_range=None, columns=None, return_type='pl'):\n    \"\"\"\n    Load data from Parquet files with optional filtering on date_id, time_id, and selected columns.\n\n    Parameters:\n    - date_id_range (tuple, optional): Range of date_id to filter (start, end). Default is None, which means all dates.\n    - time_id_range (tuple, optional): Range of time_id to filter (start, end). Default is None, which means all times.\n    - columns (list, optional): List of columns to load. Default is None, which means all columns.\n    - return_type (str, optional): Type of data to return ('pl' for Polars DataFrame or 'pd' for Pandas DataFrame). Default is 'pl'.\n\n    Returns:\n    - pl.DataFrame or pd.DataFrame: The filtered data as a Polars or Pandas DataFrame.\n    \"\"\"\n    data_dir = '../input/jane-street-real-time-market-data-forecasting'\n    # Load data using Polars lazy loading (scan_parquet)\n    data = pl.scan_parquet(f\"{data_dir}/train.parquet\")\n\n    # Apply date_id filter if specified\n    if date_id_range is not None:\n        start_date, end_date = date_id_range\n        data = data.filter((pl.col(\"date_id\") >= start_date) & (pl.col(\"date_id\") <= end_date))\n\n    # Apply time_id filter if specified\n    if time_id_range is not None:\n        start_time, end_time = time_id_range\n        data = data.filter((pl.col(\"time_id\") >= start_time) & (pl.col(\"time_id\") <= end_time))\n\n    # Select specific columns if specified\n    if columns is not None:\n        data = data.select(columns)\n\n    # Collect the data to execute the lazy operations\n    if return_type == 'pd':\n        return data.collect().to_pandas()\n    else:\n        return data.collect()\n","metadata":{"execution":{"iopub.status.busy":"2024-10-29T06:19:25.936097Z","iopub.execute_input":"2024-10-29T06:19:25.936943Z","iopub.status.idle":"2024-10-29T06:19:25.949033Z","shell.execute_reply.started":"2024-10-29T06:19:25.936853Z","shell.execute_reply":"2024-10-29T06:19:25.947875Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def calculate_r2(y_true, y_pred, weights):\n    \"\"\"\n    Calculate the sample weighted zero-mean R-squared score (R2).\n\n    Parameters:\n    - y_true (pd.Series or np.array): Ground truth values.\n    - y_pred (pd.Series or np.array): Predicted values.\n    - weights (pd.Series or np.array): Sample weights.\n\n    Returns:\n    - float: R2 score.\n    \"\"\"\n    numerator = np.sum(weights * (y_true - y_pred) ** 2)\n    denominator = np.sum(weights * (y_true ** 2))\n    r2_score = 1 - (numerator / denominator)\n    return r2_score\n\n\n\ndef evaluate_model(model, test_data):\n    y_pred = model.predict(test_data[FEAT_COLS])\n    y_true = test_data[TARGET].to_numpy() \n    weights = test_data['weight'].to_numpy()  \n\n    # Calculate R2 score\n    r2_score = calculate_r2(y_true, y_pred, weights)\n    print(f\"Sample weighted zero-mean R-squared score (R2) on test data: {r2_score}\")\n","metadata":{"execution":{"iopub.status.busy":"2024-10-29T06:19:25.952194Z","iopub.execute_input":"2024-10-29T06:19:25.952593Z","iopub.status.idle":"2024-10-29T06:19:25.960951Z","shell.execute_reply.started":"2024-10-29T06:19:25.952548Z","shell.execute_reply":"2024-10-29T06:19:25.959342Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class ModelGroup:\n    def __init__(self):\n        self.models = []\n\n    def add_model(self, model):\n        \"\"\"Add a trained model to the group.\"\"\"\n        self.models.append(model)\n\n    def predict(self, test_data):\n        \"\"\"Make predictions using all models in the group and return the average.\"\"\"\n        preds = []\n        for model in self.models:\n            if isinstance(model, lgb.Booster):\n                pred = model.predict(test_data[FEAT_COLS])\n            elif isinstance(model, xgb.Booster):\n                pred = model.predict(xgb.DMatrix(test_data[FEAT_COLS]))\n            elif hasattr(model, 'predict'):  # For PyTorch or other models\n                pred = model.predict(test_data[FEAT_COLS])\n            else:\n                raise ValueError(\"Unsupported model type\")\n            \n            preds.append(pred)\n        \n        # Average the predictions from all models\n        avg_pred = np.mean(preds, axis=0)\n        return avg_pred\n    \n    @classmethod\n    def load(cls, file_path):\n        \"\"\"Load a model group from a file.\"\"\"\n        model_group = joblib.load(file_path)\n        return model_group","metadata":{"execution":{"iopub.status.busy":"2024-10-29T06:19:25.962985Z","iopub.execute_input":"2024-10-29T06:19:25.963483Z","iopub.status.idle":"2024-10-29T06:19:25.977967Z","shell.execute_reply.started":"2024-10-29T06:19:25.963427Z","shell.execute_reply":"2024-10-29T06:19:25.976455Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def train_xgb_kfold(total_days=1498, n_splits=5, save_models=False):\n    TARGET = 'responder_6'\n    FEAT_COLS = [f\"feature_{i:02d}\" for i in range(79)]\n    \n    # 将时间序列划分为 n_splits 个部分\n    fold_size = total_days // n_splits\n    folds = [(i * fold_size, min((i + 1) * fold_size - 1, total_days - 1)) for i in range(n_splits)]\n    \n    model_group = ModelGroup()\n\n    for fold_idx in range(n_splits):\n        valid_range = folds[fold_idx]\n        train_ranges = [folds[i] for i in range(n_splits) if i != fold_idx]\n\n        print(f\"Fold {fold_idx}: validation range {valid_range}, train parts: {train_ranges}\")\n\n        valid_data = load_data(date_id_range=valid_range, columns=[\"date_id\", \"weight\"] + FEAT_COLS + [TARGET], return_type='pl')\n        valid_weight = valid_data['weight'].to_pandas()\n\n        train_data = None\n        for train_range in train_ranges:\n            partial_train_data = load_data(date_id_range=train_range, columns=[\"date_id\", \"weight\"] + FEAT_COLS + [TARGET], return_type='pl')\n            if train_data is None:\n                train_data = partial_train_data\n            else:\n                train_data = train_data.vstack(partial_train_data)\n\n        train_weight = train_data['weight'].to_pandas()\n\n        # Construct XGBoost DMatrix\n        dtrain = xgb.DMatrix(train_data.select(FEAT_COLS).to_pandas(), label=train_data[TARGET].to_pandas(), weight=train_weight)\n        dvalid = xgb.DMatrix(valid_data.select(FEAT_COLS).to_pandas(), label=valid_data[TARGET].to_pandas(), weight=valid_weight)\n\n        # Model parameters\n        XGB_PARAMS = {\n            'eval_metric': 'rmse',\n            'learning_rate': 0.5,\n            'max_depth': 6,\n            'min_child_weight': 1,\n            'subsample': 0.8,\n            'colsample_bytree': 0.8,\n            'random_state': 42,\n            'tree_method': 'gpu_hist',  \n        }\n\n        # Train model\n        model = xgb.train(\n            XGB_PARAMS,\n            dtrain,\n            num_boost_round=1000,\n            evals=[(dtrain, 'train'), (dvalid, 'valid')],\n            early_stopping_rounds=100,\n            verbose_eval=50,\n        )\n\n        # Evaluate model on validation data\n        y_valid_pred = model.predict(dvalid)\n        r2_score = calculate_r2(valid_data[TARGET].to_pandas(), y_valid_pred, valid_weight)\n        print(f\"Fold {fold_idx} validation R2 score: {r2_score}\")\n\n        # Add the model to the group\n        model_group.add_model(model)\n\n    # Optionally save the models (not required since they are in the class)\n    if save_models:\n        joblib.dump(model_group, \"xgb_model_group.pkl\")\n        print(\"Saved the model group to xgb_model_group.pkl\")\n    \n    return model_group","metadata":{"execution":{"iopub.status.busy":"2024-10-29T06:19:25.980010Z","iopub.execute_input":"2024-10-29T06:19:25.980536Z","iopub.status.idle":"2024-10-29T06:19:26.002052Z","shell.execute_reply.started":"2024-10-29T06:19:25.980483Z","shell.execute_reply":"2024-10-29T06:19:26.000419Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # 使用示例\n# total_days = 1699\n# xgb_models = train_xgb_kfold(total_days=total_days, n_splits=5, save_models=False)","metadata":{"execution":{"iopub.status.busy":"2024-10-29T06:19:26.003951Z","iopub.execute_input":"2024-10-29T06:19:26.004425Z","iopub.status.idle":"2024-10-29T06:19:26.020723Z","shell.execute_reply.started":"2024-10-29T06:19:26.004371Z","shell.execute_reply":"2024-10-29T06:19:26.019409Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 加载模型组\nxgb_models = ModelGroup.load(\"/kaggle/input/xgb_model_group_v41/other/default/1/xgb_model_group_v4.pkl\")","metadata":{"execution":{"iopub.status.busy":"2024-10-29T06:19:26.022370Z","iopub.execute_input":"2024-10-29T06:19:26.022872Z","iopub.status.idle":"2024-10-29T06:19:27.079455Z","shell.execute_reply.started":"2024-10-29T06:19:26.022822Z","shell.execute_reply":"2024-10-29T06:19:27.077994Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"lags_ : pl.DataFrame | None = None\n\n# Replace this function with your inference code.\n# You can return either a Pandas or Polars dataframe, though Polars is recommended.\n# Each batch of predictions (except the very first) must be returned within 10 minutes of the batch features being provided.\ndef predict(test: pl.DataFrame, lags: pl.DataFrame | None) -> pl.DataFrame | pd.DataFrame:\n    \"\"\"Make a prediction.\"\"\"\n    # All the responders from the previous day are passed in at time_id == 0. We save them in a global variable for access at every time_id.\n    # Use them as extra features, if you like.\n    global lags_\n    if lags is not None:\n        lags_ = lags\n\n    predictions = test.select(\n        'row_id',\n        pl.lit(0.0).alias('responder_6'),\n    )\n    \n    feat = test[FEAT_COLS].to_pandas()\n\n    pred = xgb_models.predict(feat)\n\n    \n    predictions = predictions.with_columns(pl.Series('responder_6', pred.ravel()))\n    print(predictions)\n    # The predict function must return a DataFrame\n    assert isinstance(predictions, pl.DataFrame | pd.DataFrame)\n    # with columns 'row_id', 'responer_6'\n    assert list(predictions.columns) == ['row_id', 'responder_6']\n    # and as many rows as the test data.\n    assert len(predictions) == len(test)\n\n    return predictions","metadata":{"execution":{"iopub.status.busy":"2024-10-29T06:19:27.081202Z","iopub.execute_input":"2024-10-29T06:19:27.081713Z","iopub.status.idle":"2024-10-29T06:19:27.094142Z","shell.execute_reply.started":"2024-10-29T06:19:27.081656Z","shell.execute_reply":"2024-10-29T06:19:27.092691Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"inference_server = kaggle_evaluation.jane_street_inference_server.JSInferenceServer(predict)\n\nif os.getenv('KAGGLE_IS_COMPETITION_RERUN'):\n    inference_server.serve()\nelse:\n    inference_server.run_local_gateway(\n        (\n            '/kaggle/input/jane-street-real-time-market-data-forecasting/test.parquet',\n            '/kaggle/input/jane-street-real-time-market-data-forecasting/lags.parquet',\n        )\n    )","metadata":{"execution":{"iopub.status.busy":"2024-10-29T06:19:27.095446Z","iopub.execute_input":"2024-10-29T06:19:27.095819Z","iopub.status.idle":"2024-10-29T06:19:27.584639Z","shell.execute_reply.started":"2024-10-29T06:19:27.095780Z","shell.execute_reply":"2024-10-29T06:19:27.583019Z"},"trusted":true},"execution_count":null,"outputs":[]}]}