{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":84493,"databundleVersionId":9871156,"sourceType":"competition"},{"sourceId":190073,"sourceType":"modelInstanceVersion","modelInstanceId":162056,"modelId":184419},{"sourceId":190242,"sourceType":"modelInstanceVersion","modelInstanceId":162203,"modelId":184558},{"sourceId":190267,"sourceType":"modelInstanceVersion","modelInstanceId":162223,"modelId":184579},{"sourceId":198339,"sourceType":"modelInstanceVersion","modelInstanceId":169166,"modelId":191516}],"dockerImageVersionId":30787,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"### Summary for XGBoost Version\n\nThis notebook presents an updated version of my previous LightGBM experiment, utilizing XGBoost to further explore the capabilities of different models for this competition.\n\n**1. Experiment Overview**  \nBuilding upon the LightGBM-based version, I have implemented XGBoost to compare its performance in similar settings. The key differences include changes in hyperparameters and the use of XGBoost’s GPU-accelerated training.\n\n**2. Previous LightGBM Version Summary**  \nThe previous version of the notebook focused on optimizing data loading and implementing K-fold cross-validation to ensure robust model evaluation. The complete version of that notebook is available [LGB version](https://www.kaggle.com/code/dasbro/janestreet-lgbm-dataload-kfold-baseline/notebook).\n\n**3. Future Directions**  \nMoving forward, I plan to explore ensemble methods by combining the results from different models, including LightGBM, XGBoost, and potentially others, to achieve a more robust and generalized prediction.\nSummary for XGBoost Version\nThis notebook presents an updated version of my previous LightGBM experiment, utilizing XGBoost to further explore the capabilities of different models for this competition.","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport polars as pl\nimport pandas as pd\nimport lightgbm as lgb\nimport xgboost as xgb\nimport os\nimport gc  # 引入垃圾回收模块\n\nimport kaggle_evaluation.jane_street_inference_server\n\nimport joblib","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-12-14T16:38:29.937253Z","iopub.execute_input":"2024-12-14T16:38:29.937820Z","iopub.status.idle":"2024-12-14T16:38:35.064037Z","shell.execute_reply.started":"2024-12-14T16:38:29.937779Z","shell.execute_reply":"2024-12-14T16:38:35.063346Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"xgb.__version__","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T16:38:35.065615Z","iopub.execute_input":"2024-12-14T16:38:35.066510Z","iopub.status.idle":"2024-12-14T16:38:35.072437Z","shell.execute_reply.started":"2024-12-14T16:38:35.066442Z","shell.execute_reply":"2024-12-14T16:38:35.071408Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"TARGET = 'responder_6'\nFEAT_COLS = [f\"feature_{i:02d}\" for i in range(79)]","metadata":{"execution":{"iopub.status.busy":"2024-12-14T16:38:35.073481Z","iopub.execute_input":"2024-12-14T16:38:35.073770Z","iopub.status.idle":"2024-12-14T16:38:35.087606Z","shell.execute_reply.started":"2024-12-14T16:38:35.073745Z","shell.execute_reply":"2024-12-14T16:38:35.086812Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def load_data(date_id_range=None, time_id_range=None, columns=None, return_type='pl'):\n    \"\"\"\n    Load data from Parquet files with optional filtering on date_id, time_id, and selected columns.\n\n    Parameters:\n    - date_id_range (tuple, optional): Range of date_id to filter (start, end). Default is None, which means all dates.\n    - time_id_range (tuple, optional): Range of time_id to filter (start, end). Default is None, which means all times.\n    - columns (list, optional): List of columns to load. Default is None, which means all columns.\n    - return_type (str, optional): Type of data to return ('pl' for Polars DataFrame or 'pd' for Pandas DataFrame). Default is 'pl'.\n\n    Returns:\n    - pl.DataFrame or pd.DataFrame: The filtered data as a Polars or Pandas DataFrame.\n    \"\"\"\n    data_dir = '../input/jane-street-real-time-market-data-forecasting'\n    # Load data using Polars lazy loading (scan_parquet)\n    data = pl.scan_parquet(f\"{data_dir}/train.parquet\")\n\n    # Apply date_id filter if specified\n    if date_id_range is not None:\n        start_date, end_date = date_id_range\n        data = data.filter((pl.col(\"date_id\") >= start_date) & (pl.col(\"date_id\") <= end_date))\n\n    # Apply time_id filter if specified\n    if time_id_range is not None:\n        start_time, end_time = time_id_range\n        data = data.filter((pl.col(\"time_id\") >= start_time) & (pl.col(\"time_id\") <= end_time))\n\n    # Select specific columns if specified\n    if columns is not None:\n        data = data.select(columns)\n\n    # Collect the data to execute the lazy operations\n    if return_type == 'pd':\n        return data.collect().to_pandas()\n    else:\n        return data.collect()\n","metadata":{"execution":{"iopub.status.busy":"2024-12-14T16:38:35.089830Z","iopub.execute_input":"2024-12-14T16:38:35.090141Z","iopub.status.idle":"2024-12-14T16:38:35.097013Z","shell.execute_reply.started":"2024-12-14T16:38:35.090099Z","shell.execute_reply":"2024-12-14T16:38:35.096239Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def calculate_r2(y_true, y_pred, weights):\n    \"\"\"\n    Calculate the sample weighted zero-mean R-squared score (R2).\n\n    Parameters:\n    - y_true (pd.Series or np.array): Ground truth values.\n    - y_pred (pd.Series or np.array): Predicted values.\n    - weights (pd.Series or np.array): Sample weights.\n\n    Returns:\n    - float: R2 score.\n    \"\"\"\n    numerator = np.sum(weights * (y_true - y_pred) ** 2)\n    denominator = np.sum(weights * (y_true ** 2))\n    r2_score = 1 - (numerator / denominator)\n    return r2_score\n\n\n\ndef evaluate_model(model, test_data):\n    y_pred = model.predict(test_data[FEAT_COLS])\n    y_true = test_data[TARGET].to_numpy() \n    weights = test_data['weight'].to_numpy()  \n\n    # Calculate R2 score\n    r2_score = calculate_r2(y_true, y_pred, weights)\n    print(f\"Sample weighted zero-mean R-squared score (R2) on test data: {r2_score}\")\n","metadata":{"execution":{"iopub.status.busy":"2024-12-14T16:38:35.098253Z","iopub.execute_input":"2024-12-14T16:38:35.098589Z","iopub.status.idle":"2024-12-14T16:38:35.105924Z","shell.execute_reply.started":"2024-12-14T16:38:35.098547Z","shell.execute_reply":"2024-12-14T16:38:35.105040Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**平均结果集成**","metadata":{}},{"cell_type":"code","source":"class ModelGroup:\n    def __init__(self):\n        self.models = []\n    \n    def add_model(self, model):\n        self.models.append(model)\n\n    def load_models(self, filepath):\n        self.models = joblib.load(filepath)\n        \n    def predict(self, data, predict_disable_shape_check=True):\n        # 对每个模型进行预测并返回平均结果\n        predictions = np.zeros((len(data), len(self.models)))\n        for i, model in enumerate(self.models):\n            if isinstance(model, xgb.Booster):\n                # XGBoost模型预测\n                ddata = xgb.DMatrix(data)\n                predictions[:, i] = model.predict(ddata)\n            elif isinstance(model, lgb.Booster):\n                # LightGBM模型预测\n                predictions[:, i] = model.predict(data,predict_disable_shape_check=True)\n        return predictions.mean(axis=1)\n\n    @classmethod\n    def load(cls, file_path):\n        \"\"\"Load a model group from a file.\"\"\"\n        model_group = joblib.load(file_path)\n        return model_group\n    \n\ndef train_and_ensemble(total_days=1498, n_splits=5, save_models=False):\n    # 使用LightGBM训练模型\n    lgb_model_group = train_lgb_kfold_incremental(total_days=total_days, n_splits=n_splits, save_models=save_models)\n    # 使用XGBoost训练模型\n    xgb_model_group = train_xgb_kfold_incremental(total_days=total_days, n_splits=n_splits, save_models=save_models)\n    \n    \n    \n    # 集成模型\n    ensemble_model_group = ModelGroup()\n    ensemble_model_group.models = xgb_model_group.models + lgb_model_group.models\n    \n    # 可以选择保存集成后的模型\n    if save_models:\n        joblib.dump(ensemble_model_group, \"ensemble_model_group.pkl\")\n        print(\"Saved the ensemble model group to ensemble_model_group.pkl\")\n    \n    return ensemble_model_group\n\n# 预测新数据\ndef predict_with_ensemble(ensemble_model_group, new_data, predict_disable_shape_check=True):\n    # 使用集成模型对新数据进行预测\n    return ensemble_model_group.predict(new_data,predict_with_ensemble)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T16:38:35.106918Z","iopub.execute_input":"2024-12-14T16:38:35.107185Z","iopub.status.idle":"2024-12-14T16:38:35.117640Z","shell.execute_reply.started":"2024-12-14T16:38:35.107149Z","shell.execute_reply":"2024-12-14T16:38:35.116872Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 加载LightGBM和XGBoost模型的pkl文件并进行集成\ndef load_and_ensemble(lgb_model_path, xgb_model_path, save_models=False):\n    # 加载LightGBM和XGBoost的模型\n    lgb_model_group = joblib.load(lgb_model_path)\n    xgb_model_group = joblib.load(xgb_model_path)\n    print(type(lgb_model_group))\n\n    # 集成模型\n    ensemble_model_group = ModelGroup()\n    ensemble_model_group.models = xgb_model_group.models + lgb_model_group.models\n    \n    # 可以选择保存集成后的模型\n    if save_models:\n        joblib.dump(ensemble_model_group, \"ensemble_model_group.pkl\")\n        print(\"Saved the ensemble model group to ensemble_model_group.pkl\")\n    \n    return ensemble_model_group\n\n# 使用示例\nlgb_model_path = \"/kaggle/input/lgbm_incremental/pytorch/default/1/lgb_model_group_incremental (2).pkl\"\nxgb_model_path = \"/kaggle/input/xgb_model/other/default/1/xgb_model_group.pkl\"\n\n# 加载模型并进行集成\n# ensemble_model_group = load_and_ensemble(lgb_model_path, xgb_model_path, save_models=True)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T16:38:35.118681Z","iopub.execute_input":"2024-12-14T16:38:35.118924Z","iopub.status.idle":"2024-12-14T16:38:35.614328Z","shell.execute_reply.started":"2024-12-14T16:38:35.118896Z","shell.execute_reply":"2024-12-14T16:38:35.612446Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**XGBoost**","metadata":{}},{"cell_type":"code","source":"def train_xgb_kfold_incremental(total_days=1498, n_splits=5, save_models=False):\n    TARGET = 'responder_6'\n    FEAT_COLS = [f\"feature_{i:02d}\" for i in range(78)]\n\n    fold_size = total_days // n_splits\n    folds = [(i * fold_size, min((i + 1) * fold_size - 1, total_days - 1)) for i in range(n_splits)]\n\n    model_group = ModelGroup()\n    r2_scores = []\n\n    for fold_idx in range(n_splits):\n        valid_range = folds[fold_idx]\n        train_ranges = [folds[i] for i in range(n_splits) if i != fold_idx]\n\n        print(f\"Fold {fold_idx}: validation range {valid_range}, train parts: {train_ranges}\")\n\n        # 加载验证集\n        valid_data = load_data(date_id_range=valid_range, columns=[\"date_id\", \"weight\"] + FEAT_COLS + [TARGET], return_type='pl')\n        valid_weight = valid_data['weight'].to_pandas()\n\n        dvalid = xgb.DMatrix(valid_data.select(FEAT_COLS).to_pandas(), label=valid_data[TARGET].to_pandas(), weight=valid_weight)\n\n        # 初始化 XGBoost 参数\n        XGB_PARAMS = {\n            'objective': 'reg:squarederror',\n            'eval_metric': 'rmse',\n            'learning_rate': 0.1,\n            'max_depth': 6,\n            'min_child_weight': 1,\n            'subsample': 0.8,\n            'colsample_bytree': 0.8,\n            'random_state': 42,\n            'tree_method': 'gpu_hist',\n        }\n\n        # 初始化模型\n        model = None\n        evals_result = {}\n        iteration = 0  # 初始化迭代计数\n\n        for train_range in train_ranges:\n            print(f\"Training with range {train_range}...\")\n\n            # 加载当前训练数据块\n            partial_train_data = load_data(date_id_range=train_range, columns=[\"date_id\", \"weight\"] + FEAT_COLS + [TARGET], return_type='pl')\n            train_weight = partial_train_data['weight'].to_pandas()\n            dtrain = xgb.DMatrix(partial_train_data.select(FEAT_COLS).to_pandas(), label=partial_train_data[TARGET].to_pandas(), weight=train_weight)\n\n            # 如果是第一次训练，创建 Booster\n            if model is None:\n                model = xgb.train(\n                    XGB_PARAMS,\n                    dtrain,\n                    num_boost_round=100,\n                    evals=[(dtrain, 'train')],\n                    evals_result=evals_result,\n                    verbose_eval=False\n                )\n            else:\n                # 增量更新\n                model.update(dtrain, iteration)\n\n            # 增加迭代次数\n            iteration += 1\n\n            # 清理训练数据块和 DMatrix 对象\n            del partial_train_data\n            del dtrain\n            gc.collect()  # 强制垃圾回收\n\n        # 验证模型\n        y_valid_pred = model.predict(dvalid)\n        r2_score = calculate_r2(valid_data[TARGET].to_pandas(), y_valid_pred, valid_weight)\n        print(f\"Fold {fold_idx} validation R2 score: {r2_score}\")\n        r2_scores.append(r2_score)\n\n        # 保存模型\n        model_group.add_model(model)\n\n        # 清理验证数据块和 DMatrix 对象\n        del valid_data\n        del dvalid\n        gc.collect()  # 强制垃圾回收\n\n    if save_models:\n        joblib.dump(model_group, \"xgb_model_group_incremental.pkl\")\n        print(\"Saved the XGBoost incremental model group to xgb_model_group_incremental.pkl\")\n\n    return model_group\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T16:38:35.615224Z","iopub.execute_input":"2024-12-14T16:38:35.615547Z","iopub.status.idle":"2024-12-14T16:38:35.629287Z","shell.execute_reply.started":"2024-12-14T16:38:35.615492Z","shell.execute_reply":"2024-12-14T16:38:35.628622Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**LGBM**","metadata":{}},{"cell_type":"code","source":"import lightgbm as lgb\nimport pandas as pd\nimport joblib\nimport gc  # For memory management\n\n# Helper function to calculate R2 score\ndef calculate_r2(y_true, y_pred, weights):\n    numerator = ((weights * (y_true - y_pred) ** 2).sum())\n    denominator = (weights * (y_true ** 2)).sum()\n    return 1 - numerator / denominator\n\n# Main training function with DMatrix and incremental updates\ndef train_lgb_kfold_incremental(total_days=1498, n_splits=5, save_models=False):\n    # Split days into folds\n    fold_size = total_days // n_splits\n    folds = [(i * fold_size, min((i + 1) * fold_size - 1, total_days - 1)) for i in range(n_splits)]\n    \n    model_group = ModelGroup()\n\n    for fold_idx in range(n_splits):\n        valid_range = folds[fold_idx]\n        train_ranges = [folds[i] for i in range(n_splits) if i != fold_idx]\n\n        print(f\"Fold {fold_idx}: validation range {valid_range}, train parts: {train_ranges}\")\n\n        # Load validation data\n        valid_data = load_data(date_id_range=valid_range, columns=[\"date_id\", \"weight\"] + FEAT_COLS + [TARGET], return_type='pl')\n        valid_weight = valid_data['weight'].to_pandas()\n        valid_dmatrix = lgb.Dataset(\n            data=valid_data.select(FEAT_COLS).to_pandas(),\n            label=valid_data[TARGET].to_pandas(),\n            weight=valid_weight,\n            free_raw_data=False  # Prevent raw data from being freed\n        )\n\n        # Initialize LightGBM parameters\n        LGB_PARAMS = {\n            'objective': 'regression_l2',\n            'metric': 'rmse',\n            'learning_rate': 0.05,\n            'num_leaves': 31,\n            'max_depth': -1,\n            'random_state': 42,\n            'device': 'gpu',\n        }\n\n        early_stopping_callback = lgb.early_stopping(100)\n        verbose_eval_callback = lgb.log_evaluation(period=50)\n\n        # Initialize model to None\n        model = None\n\n        # Load and train data incrementally\n        for train_range in train_ranges:\n            print(f\"Training with range {train_range}...\")\n\n            # Load current training batch\n            partial_train_data = load_data(date_id_range=train_range, columns=[\"date_id\", \"weight\"] + FEAT_COLS + [TARGET], return_type='pl')\n            train_weight = partial_train_data['weight'].to_pandas()\n            train_dmatrix = lgb.Dataset(\n                data=partial_train_data.select(FEAT_COLS).to_pandas(),\n                label=partial_train_data[TARGET].to_pandas(),\n                weight=train_weight,\n                free_raw_data=False  # Prevent raw data from being freed\n            )\n\n            # If model is None, train a new model\n            if model is None:\n                model = lgb.train(\n                    LGB_PARAMS,\n                    train_dmatrix,\n                    num_boost_round=1000,\n                    valid_sets=[train_dmatrix, valid_dmatrix],\n                    valid_names=['train', 'valid'],\n                    callbacks=[early_stopping_callback, verbose_eval_callback]\n                )\n            else:\n                # Continue training the existing model (incremental update)\n                model = lgb.train(\n                    LGB_PARAMS,\n                    train_dmatrix,\n                    num_boost_round=1000,\n                    valid_sets=[train_dmatrix, valid_dmatrix],\n                    valid_names=['train', 'valid'],\n                    callbacks=[early_stopping_callback, verbose_eval_callback],\n                    init_model=model  # Incremental update by using the existing model\n                )\n\n            # Free memory after training each batch\n            del partial_train_data, train_dmatrix\n            gc.collect()\n\n        # Evaluate model on validation data\n        y_valid_pred = model.predict(valid_data.select(FEAT_COLS).to_pandas())\n        r2_score = calculate_r2(valid_data[TARGET].to_pandas(), y_valid_pred, valid_weight)\n        print(f\"Fold {fold_idx} validation R2 score: {r2_score}\")\n\n        # Add the model to the group\n        model_group.add_model(model)\n\n        # Free memory explicitly\n        del model, valid_dmatrix, valid_data\n        gc.collect()\n\n    # Optionally save the models\n    if save_models:\n        joblib.dump(model_group, \"lgb_model_group_incremental.pkl\")\n        print(\"Saved the model group to lgb_model_group_incremental.pkl\")\n\n    return model_group\n\n# Usage\n# Assuming load_data, FEAT_COLS, and TARGET are defined.\n# train_lgb_kfold_incremental()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T16:38:35.630435Z","iopub.execute_input":"2024-12-14T16:38:35.630775Z","iopub.status.idle":"2024-12-14T16:38:35.764219Z","shell.execute_reply.started":"2024-12-14T16:38:35.630744Z","shell.execute_reply":"2024-12-14T16:38:35.763446Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ensemble_model = train_and_ensemble(total_days=1498, n_splits=5, save_models=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T16:38:35.767124Z","iopub.execute_input":"2024-12-14T16:38:35.767640Z","iopub.status.idle":"2024-12-14T16:38:35.777616Z","shell.execute_reply.started":"2024-12-14T16:38:35.767606Z","shell.execute_reply":"2024-12-14T16:38:35.776787Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# total_days = 69\n# lgb_models = train_lgb_kfold(total_days=total_days,n_splits =5)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T16:38:35.778389Z","iopub.execute_input":"2024-12-14T16:38:35.778667Z","iopub.status.idle":"2024-12-14T16:38:35.787179Z","shell.execute_reply.started":"2024-12-14T16:38:35.778638Z","shell.execute_reply":"2024-12-14T16:38:35.786510Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 使用示例\n# total_days = 69\n# xgb_models = train_xgb_kfold(total_days=total_days, n_splits=5, save_models=False)","metadata":{"execution":{"iopub.status.busy":"2024-12-14T16:38:35.790192Z","iopub.execute_input":"2024-12-14T16:38:35.790450Z","iopub.status.idle":"2024-12-14T16:38:35.796885Z","shell.execute_reply.started":"2024-12-14T16:38:35.790422Z","shell.execute_reply":"2024-12-14T16:38:35.796206Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"TARGET = 'responder_6'\nFEAT_COLS = [f\"feature_{i:02d}\" for i in range(79)]","metadata":{"execution":{"iopub.status.busy":"2024-12-14T16:38:35.798042Z","iopub.execute_input":"2024-12-14T16:38:35.801730Z","iopub.status.idle":"2024-12-14T16:38:35.805566Z","shell.execute_reply.started":"2024-12-14T16:38:35.801694Z","shell.execute_reply":"2024-12-14T16:38:35.804804Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"ensemble_models = ModelGroup.load(\"/kaggle/input/esemble_1/other/default/1/ensemble_model_group (3).pkl\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T16:38:35.806854Z","iopub.execute_input":"2024-12-14T16:38:35.807160Z","iopub.status.idle":"2024-12-14T16:38:36.040756Z","shell.execute_reply.started":"2024-12-14T16:38:35.807126Z","shell.execute_reply":"2024-12-14T16:38:36.039886Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"lags_ : pl.DataFrame | None = None\n\n# Replace this function with your inference code.\n# You can return either a Pandas or Polars dataframe, though Polars is recommended.\n# Each batch of predictions (except the very first) must be returned within 10 minutes of the batch features being provided.\ndef predict(test: pl.DataFrame, lags: pl.DataFrame | None) -> pl.DataFrame | pd.DataFrame:\n    \"\"\"Make a prediction.\"\"\"\n    # All the responders from the previous day are passed in at time_id == 0. We save them in a global variable for access at every time_id.\n    # Use them as extra features, if you like.\n    global lags_\n    if lags is not None:\n        lags_ = lags\n\n    predictions = test.select(\n        'row_id',\n        pl.lit(0.0).alias('responder_6'),\n    )\n    feat = test[FEAT_COLS].to_pandas()\n\n\n    pred = predict_with_ensemble(ensemble_models,feat)\n\n    \n    predictions = predictions.with_columns(pl.Series('responder_6', pred.ravel()))\n    print(predictions)\n    # The predict function must return a DataFrame\n    assert isinstance(predictions, pl.DataFrame | pd.DataFrame)\n    # with columns 'row_id', 'responer_6'\n    assert list(predictions.columns) == ['row_id', 'responder_6']\n    # and as many rows as the test data.\n    assert len(predictions) == len(test)\n\n    return predictions","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T16:38:36.041661Z","iopub.execute_input":"2024-12-14T16:38:36.043671Z","iopub.status.idle":"2024-12-14T16:38:36.054630Z","shell.execute_reply.started":"2024-12-14T16:38:36.043632Z","shell.execute_reply":"2024-12-14T16:38:36.053869Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"inference_server = kaggle_evaluation.jane_street_inference_server.JSInferenceServer(predict)\n\nif os.getenv('KAGGLE_IS_COMPETITION_RERUN'):\n    inference_server.serve()\nelse:\n    inference_server.run_local_gateway(\n        (\n            '/kaggle/input/jane-street-real-time-market-data-forecasting/test.parquet',\n            '/kaggle/input/jane-street-real-time-market-data-forecasting/lags.parquet',\n        )\n    )","metadata":{"execution":{"iopub.status.busy":"2024-12-14T16:38:36.055586Z","iopub.execute_input":"2024-12-14T16:38:36.056817Z","iopub.status.idle":"2024-12-14T16:38:36.419372Z","shell.execute_reply.started":"2024-12-14T16:38:36.056783Z","shell.execute_reply":"2024-12-14T16:38:36.417115Z"},"trusted":true},"outputs":[],"execution_count":null}]}