{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"codemirror_mode":{"name":"ipython","version":3},"file_extension":".py","mimetype":"text/x-python","name":"python","nbconvert_exporter":"python","pygments_lexer":"ipython3","version":"3.10.14"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":84493,"databundleVersionId":9871156,"sourceType":"competition"},{"sourceId":9801075,"sourceType":"datasetVersion","datasetId":6006872},{"sourceId":9806342,"sourceType":"datasetVersion","datasetId":6010899},{"sourceId":203900450,"sourceType":"kernelVersion"}],"dockerImageVersionId":30787,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true},"papermill":{"default_parameters":{},"duration":7.594014,"end_time":"2024-10-10T11:58:36.355301","environment_variables":{},"exception":null,"input_path":"__notebook__.ipynb","output_path":"__notebook__.ipynb","parameters":{},"start_time":"2024-10-10T11:58:28.761287","version":"2.6.0"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"Here's the translation:\n\n\"I am not the original author of the code; I have simply added detailed comments to this code to make it easier to understand.\"","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport polars as pl\nimport numpy as np\nimport os, gc\nfrom tqdm.auto import tqdm\nfrom matplotlib import pyplot as plt\nimport pickle\n\nimport torch\nimport torch.nn as nn\nimport torch.nn.functional as F\nfrom pytorch_lightning import (LightningDataModule, LightningModule, Trainer)\nfrom pytorch_lightning.callbacks import EarlyStopping, ModelCheckpoint, Timer\n\nimport pandas as pd\nimport numpy as np\nfrom sklearn.metrics import r2_score\nfrom sklearn.model_selection import train_test_split\nfrom torch.utils.data import Dataset, DataLoader\n\n\nfrom sklearn.metrics import r2_score\nfrom lightgbm import LGBMRegressor\nimport lightgbm as lgb\nfrom xgboost import XGBRegressor\nfrom catboost import CatBoostRegressor\nfrom sklearn.ensemble import VotingRegressor\n\nimport warnings\nwarnings.filterwarnings('ignore')\npd.options.display.max_columns = None\n\nimport kaggle_evaluation.jane_street_inference_server","metadata":{"execution":{"iopub.status.busy":"2024-11-05T04:58:17.952915Z","iopub.execute_input":"2024-11-05T04:58:17.953615Z","iopub.status.idle":"2024-11-05T04:58:17.962711Z","shell.execute_reply.started":"2024-11-05T04:58:17.953574Z","shell.execute_reply":"2024-11-05T04:58:17.961348Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# NN + XGB inference","metadata":{}},{"cell_type":"markdown","source":"# Configurations","metadata":{}},{"cell_type":"code","source":"class CONFIG:\n    # 设置随机种子，确保实验结果的可复现性\n    # Set random seed for reproducibility of results\n    seed = 42\n\n    # 目标列（模型预测的目标变量）\n    # The target column (the variable that the model will predict)\n    target_col = \"responder_6\"\n\n    # 特征列：包括79个 \"feature_xx\" 特征列和9个滞后1期的 \"responder_xx_lag_1\" 特征列\n    # Feature columns: consists of 79 \"feature_xx\" columns and 9 lag-1 \"responder_xx_lag_1\" columns\n    feature_cols = [f\"feature_{idx:02d}\" for idx in range(79)] + [f\"responder_{idx}_lag_1\" for idx in range(9)]\n    \n    # 模型路径列表：包含多个训练好的模型文件路径\n    # Model paths: list of paths to pre-trained models\n    model_paths = [\n        #\"/kaggle/input/js24-train-gbdt-model-with-lags-singlemodel/result.pkl\",  # 示例路径，已注释掉\n        #\"/kaggle/input/js24-trained-gbdt-model/result.pkl\",  # 示例路径，已注释掉\n        \"/kaggle/input/js-xs-nn-trained-model\",  # 预训练神经网络模型路径\n        \"/kaggle/input/js-with-lags-trained-xgb/result.pkl\",  # 预训练的XGBoost模型路径\n    ]\n","metadata":{"execution":{"iopub.status.busy":"2024-11-05T04:58:17.964754Z","iopub.execute_input":"2024-11-05T04:58:17.965474Z","iopub.status.idle":"2024-11-05T04:58:17.993759Z","shell.execute_reply.started":"2024-11-05T04:58:17.965427Z","shell.execute_reply":"2024-11-05T04:58:17.992851Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Load preprocessed data (to calculate CV)","metadata":{}},{"cell_type":"code","source":"valid = pl.scan_parquet(\n    f\"/kaggle/input/js24-preprocessing-create-lags/validation.parquet/\"\n).collect().to_pandas()","metadata":{"execution":{"iopub.status.busy":"2024-11-05T04:58:17.994912Z","iopub.execute_input":"2024-11-05T04:58:17.995267Z","iopub.status.idle":"2024-11-05T04:58:18.776454Z","shell.execute_reply.started":"2024-11-05T04:58:17.995223Z","shell.execute_reply":"2024-11-05T04:58:18.77564Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Load model","metadata":{}},{"cell_type":"code","source":"# 初始化 xgb_model 变量为 None\n# Initialize the `xgb_model` variable to None\nxgb_model = None\n\n# 从 CONFIG 配置中获取模型路径列表，选择第一个 XGBoost 模型路径\n# Get the model path from the CONFIG configuration, select the second XGBoost model path (index 1)\nmodel_path = CONFIG.model_paths[1]\n\n# 打开模型文件，并加载模型\n# Open the model file and load the model\nwith open(model_path, \"rb\") as fp:\n    # 使用 pickle 加载模型文件中的内容\n    # Use pickle to load the contents of the model file\n    result = pickle.load(fp)\n    # 从加载的结果中提取模型对象，保存在 xgb_model 变量中\n    # Extract the model object from the loaded result and store it in `xgb_model`\n    xgb_model = result[\"model\"]\n\n# 定义模型输入特征列（包含 \"symbol_id\", \"time_id\" 和来自 CONFIG 配置的特征列）\n# Define the feature columns for the model, including \"symbol_id\", \"time_id\", and the feature columns from the CONFIG configuration\nxgb_feature_cols = [\"symbol_id\", \"time_id\"] + CONFIG.feature_cols\n\n# 显示加载的 XGBoost 模型\n# Display the loaded XGBoost model\ndisplay(xgb_model)\n","metadata":{"execution":{"iopub.status.busy":"2024-11-05T04:58:18.778669Z","iopub.execute_input":"2024-11-05T04:58:18.778953Z","iopub.status.idle":"2024-11-05T04:58:18.796672Z","shell.execute_reply.started":"2024-11-05T04:58:18.778923Z","shell.execute_reply":"2024-11-05T04:58:18.795818Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 自定义 R2 指标，用于验证阶段\n# Custom R2 metric for validation\ndef r2_val(y_true, y_pred, sample_weight):\n    # 计算加权 R2 值，避免除以 0 的情况使用 1e-38\n    # Calculate weighted R2 value, with 1e-38 to avoid division by zero\n    r2 = 1 - np.average((y_pred - y_true) ** 2, weights=sample_weight) / (np.average((y_true) ** 2, weights=sample_weight) + 1e-38)\n    return r2\n\n\nclass NN(LightningModule):\n    # 初始化神经网络模型\n    # Initialize the neural network model\n    def __init__(self, input_dim, hidden_dims, dropouts, lr, weight_decay):\n        super().__init__()\n        # 保存超参数，方便在训练时使用\n        # Save hyperparameters for use in training\n        self.save_hyperparameters()\n        \n        layers = []\n        in_dim = input_dim\n        \n        # 构建隐藏层，逐层添加 BatchNorm, 激活函数, Dropout 和 Linear 层\n        # Construct the hidden layers, adding BatchNorm, activation function, Dropout, and Linear layers step by step\n        for i, hidden_dim in enumerate(hidden_dims):\n            layers.append(nn.BatchNorm1d(in_dim))  # 添加批归一化层\n            if i > 0:\n                layers.append(nn.SiLU())  # 激活函数 SiLU (Sigmoid Linear Unit)\n            if i < len(dropouts):\n                layers.append(nn.Dropout(dropouts[i]))  # Dropout层，避免过拟合\n            layers.append(nn.Linear(in_dim, hidden_dim))  # 全连接层（线性变换）\n            in_dim = hidden_dim  # 更新输入维度\n            \n        layers.append(nn.Linear(in_dim, 1))  # 输出层，最后一层是线性层\n        layers.append(nn.Tanh())  # 使用 Tanh 激活函数来归一化输出值\n        self.model = nn.Sequential(*layers)  # 将所有层组成一个顺序模型\n        \n        self.lr = lr  # 学习率\n        self.weight_decay = weight_decay  # 权重衰减（L2 正则化）\n        self.validation_step_outputs = []  # 用于保存每个验证步骤的输出\n\n    # 前向传播函数\n    # Forward pass function\n    def forward(self, x):\n        return 5 * self.model(x).squeeze(-1)  # 输出通过模型，去除多余的维度并乘以5（缩放因子）\n\n    # 训练步骤\n    # Training step\n    def training_step(self, batch):\n        x, y, w = batch  # 获取输入数据 x，标签 y 和样本权重 w\n        y_hat = self(x)  # 使用模型进行预测\n        loss = F.mse_loss(y_hat, y, reduction='none') * w  # 计算加权的 MSE 损失\n        loss = loss.mean()  # 对所有样本的损失进行平均\n        self.log('train_loss', loss, on_step=False, on_epoch=True, batch_size=x.size(0))  # 记录训练损失\n        return loss  # 返回损失值\n\n    # 验证步骤\n    # Validation step\n    def validation_step(self, batch):\n        x, y, w = batch  # 获取输入数据 x，标签 y 和样本权重 w\n        y_hat = self(x)  # 使用模型进行预测\n        loss = F.mse_loss(y_hat, y, reduction='none') * w  # 计算加权的 MSE 损失\n        loss = loss.mean()  # 对所有样本的损失进行平均\n        self.log('val_loss', loss, on_step=False, on_epoch=True, batch_size=x.size(0))  # 记录验证损失\n        self.validation_step_outputs.append((y_hat, y, w))  # 保存每个步骤的输出，用于计算其他指标\n        return loss  # 返回损失值\n\n    # 每个验证周期结束时调用，用于计算验证集的 R2 值\n    # Called at the end of each validation epoch to calculate R2 on the validation set\n    def on_validation_epoch_end(self):\n        \"\"\"Calculate validation WRMSE at the end of the epoch.\"\"\"\n        y = torch.cat([x[1] for x in self.validation_step_outputs]).cpu().numpy()  # 获取所有标签值\n        if self.trainer.sanity_checking:\n            prob = torch.cat([x[0] for x in self.validation_step_outputs]).cpu().numpy()  # 如果是验证步骤，直接获取预测值\n        else:\n            prob = torch.cat([x[0] for x in self.validation_step_outputs]).cpu().numpy()  # 获取所有预测值\n            weights = torch.cat([x[2] for x in self.validation_step_outputs]).cpu().numpy()  # 获取所有权重\n            # 计算加权的 R2 值\n            val_r_square = r2_val(y, prob, weights)\n            self.log(\"val_r_square\", val_r_square, prog_bar=True, on_step=False, on_epoch=True)  # 记录验证集的 R2 值\n        self.validation_step_outputs.clear()  # 清空步骤输出，为下一个周期准备\n\n    # 配置优化器和学习率调度器\n    # Configure the optimizer and learning rate scheduler\n    def configure_optimizers(self):\n        # 使用 Adam 优化器，包含学习率和权重衰减（L2 正则化）\n        optimizer = torch.optim.Adam(self.parameters(), lr=self.lr, weight_decay=self.weight_decay)\n        # 使用 ReduceLROnPlateau 学习率调度器，在验证损失不下降时减少学习率\n        scheduler = torch.optim.lr_scheduler.ReduceLROnPlateau(optimizer, mode='min', factor=0.5, patience=5,\n                                                               verbose=True)\n        return {\n            'optimizer': optimizer,  # 返回优化器\n            'lr_scheduler': {\n                'scheduler': scheduler,  # 返回学习率调度器\n                'monitor': 'val_loss',  # 监控验证损失以调整学习率\n            }\n        }\n\n    # 每个训练周期结束时调用，用于打印训练过程中的指标\n    # Called at the end of each training epoch to print metrics\n    def on_train_epoch_end(self):\n        if self.trainer.sanity_checking:\n            return  # 如果是验证步骤，跳过输出\n        epoch = self.trainer.current_epoch  # 当前训练周期\n        # 获取训练过程中的所有指标\n        metrics = {k: v.item() if isinstance(v, torch.Tensor) else v for k, v in self.trainer.logged_metrics.items()}\n        # 格式化输出指标为小数点后5位\n        formatted_metrics = {k: f\"{v:.5f}\" for k, v in metrics.items()}\n        # 打印当前训练周期的指标\n        print(f\"Epoch {epoch}: {formatted_metrics}\")\n","metadata":{"execution":{"iopub.status.busy":"2024-11-05T04:58:18.797992Z","iopub.execute_input":"2024-11-05T04:58:18.798262Z","iopub.status.idle":"2024-11-05T04:58:18.818565Z","shell.execute_reply.started":"2024-11-05T04:58:18.798232Z","shell.execute_reply":"2024-11-05T04:58:18.817654Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 定义交叉验证的折数（5折交叉验证）\n# Set the number of folds for cross-validation (5-fold cross-validation)\nN_folds = 5\n\n# 初始化一个空列表，用于保存训练好的模型\n# Initialize an empty list to store the trained models\nmodels = []\n\n# 循环加载每个折数的最佳模型\n# Loop through each fold and load the best model\nfor fold in range(N_folds):\n    # 构造每个折数对应模型的 checkpoint 路径\n    # Construct the checkpoint path for each fold model\n    checkpoint_path = f\"{CONFIG.model_paths[0]}/nn_{fold}.model\"\n    \n    # 加载模型（从 checkpoint 文件中恢复）\n    # Load the model from the checkpoint file\n    model = NN.load_from_checkpoint(checkpoint_path)\n    \n    # 将模型移到 GPU 上（假设使用 CUDA 设备 0）\n    # Move the model to the GPU (assumed to be CUDA device 0)\n    models.append(model.to(\"cuda:0\"))\n","metadata":{"execution":{"iopub.status.busy":"2024-11-05T04:58:18.819804Z","iopub.execute_input":"2024-11-05T04:58:18.820186Z","iopub.status.idle":"2024-11-05T04:58:19.096635Z","shell.execute_reply.started":"2024-11-05T04:58:18.820154Z","shell.execute_reply":"2024-11-05T04:58:19.095845Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# CV Score","metadata":{}},{"cell_type":"code","source":"# 从验证集提取特征列 X_valid 和目标变量 y_valid\n# Extract feature columns X_valid and target variable y_valid from the validation set\nX_valid = valid[xgb_feature_cols]  # 提取特征列\ny_valid = valid[CONFIG.target_col]  # 提取目标列（标签）\n\n# 提取验证集中的样本权重 w_valid\n# Extract sample weights from the validation set\nw_valid = valid[\"weight\"]\n\n# 使用 XGBoost 模型对验证集进行预测\n# Predict the validation set using the XGBoost model\ny_pred_valid_xgb = xgb_model.predict(X_valid)\n\n# 计算加权 R2 分数（验证集的 R2 值）\n# Calculate the weighted R2 score (validation R2 score)\nvalid_score = r2_score(y_valid, y_pred_valid_xgb, sample_weight=w_valid)\n\n# 输出验证集的 R2 分数\n# Output the validation R2 score\nvalid_score\n","metadata":{"execution":{"iopub.status.busy":"2024-11-05T04:58:19.097685Z","iopub.execute_input":"2024-11-05T04:58:19.099563Z","iopub.status.idle":"2024-11-05T04:58:21.925894Z","shell.execute_reply.started":"2024-11-05T04:58:19.099528Z","shell.execute_reply":"2024-11-05T04:58:21.924796Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 从验证集提取特征列 X_valid 和目标变量 y_valid\n# Extract feature columns X_valid and target variable y_valid from the validation set\nX_valid = valid[CONFIG.feature_cols]  # 提取特征列（使用配置中的 feature_cols 列表）\ny_valid = valid[CONFIG.target_col]  # 提取目标列（标签）\n\n# 提取验证集中的样本权重 w_valid\n# Extract sample weights from the validation set\nw_valid = valid[\"weight\"]\n\n# 对 X_valid 中的缺失值进行填充：\n# 1. 使用前向填充方法（'ffill'）填充缺失值\n# 2. 如果仍有缺失值，使用 0 填充\n# Fill missing values in X_valid:\n# 1. Forward fill ('ffill') to fill missing values.\n# 2. If still missing, fill with 0.\nX_valid = X_valid.fillna(method='ffill').fillna(0)\n\n# 输出 X_valid, y_valid 和 w_valid 的形状\n# Output the shapes of X_valid, y_valid, and w_valid\nX_valid.shape, y_valid.shape, w_valid.shape\n","metadata":{"execution":{"iopub.status.busy":"2024-11-05T04:58:21.927553Z","iopub.execute_input":"2024-11-05T04:58:21.927978Z","iopub.status.idle":"2024-11-05T04:58:23.887875Z","shell.execute_reply.started":"2024-11-05T04:58:21.927931Z","shell.execute_reply":"2024-11-05T04:58:23.88685Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 初始化一个全零的数组 y_pred_valid_nn，用于存储神经网络模型的预测结果\n# Initialize an array of zeros, y_pred_valid_nn, to store predictions from the neural network models\ny_pred_valid_nn = np.zeros(y_valid.shape)\n\n# 在不计算梯度的情况下进行预测\n# Make predictions without calculating gradients (inference mode)\nwith torch.no_grad():\n    for model in models:\n        model.eval()  # 设置模型为评估模式（禁用 Dropout 和 BatchNorm）\n        # 将验证集特征数据转换为 FloatTensor，并移到 GPU 上\n        # Convert validation features to FloatTensor and move them to GPU\n        y_pred_valid_nn += model(torch.FloatTensor(X_valid.values).to(\"cuda:0\")).cpu().numpy() / len(models)\n        # 对每个模型的预测结果进行加权平均\n        # Take the average of predictions from all models\n\n# 计算加权 R2 分数（验证集的 R2 值）\n# Calculate the weighted R2 score (validation R2 score)\nvalid_score = r2_score(y_valid, y_pred_valid_nn, sample_weight=w_valid)\n\n# 输出验证集的 R2 分数\n# Output the validation R2 score\nvalid_score\n","metadata":{"execution":{"iopub.status.busy":"2024-11-05T04:58:23.889005Z","iopub.execute_input":"2024-11-05T04:58:23.889319Z","iopub.status.idle":"2024-11-05T04:58:30.150346Z","shell.execute_reply.started":"2024-11-05T04:58:23.889286Z","shell.execute_reply":"2024-11-05T04:58:30.149345Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 使用 XGBoost 和神经网络模型的加权平均作为集成预测结果\n# Combine XGBoost and neural network model predictions with weighted average\ny_pred_valid_ensemble = 0.5 * (y_pred_valid_xgb + y_pred_valid_nn)\n\n# 计算加权 R2 分数（集成模型的 R2 值）\n# Calculate the weighted R2 score (R2 score for the ensemble model)\nvalid_score = r2_score(y_valid, y_pred_valid_ensemble, sample_weight=w_valid)\n\n# 输出集成模型的 R2 分数\n# Output the R2 score for the ensemble model\nvalid_score\n","metadata":{"execution":{"iopub.status.busy":"2024-11-05T04:58:30.153827Z","iopub.execute_input":"2024-11-05T04:58:30.154384Z","iopub.status.idle":"2024-11-05T04:58:30.17176Z","shell.execute_reply.started":"2024-11-05T04:58:30.154344Z","shell.execute_reply":"2024-11-05T04:58:30.170735Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 删除不再使用的变量，以释放内存\n# Delete variables that are no longer needed to free up memory\ndel valid, X_valid, y_valid, w_valid\n\n# 手动触发垃圾回收，释放未使用的内存\n# Explicitly trigger garbage collection to release unused memory\ngc.collect()\n","metadata":{"execution":{"iopub.status.busy":"2024-11-05T04:58:30.1728Z","iopub.execute_input":"2024-11-05T04:58:30.173077Z","iopub.status.idle":"2024-11-05T04:58:30.386612Z","shell.execute_reply.started":"2024-11-05T04:58:30.173047Z","shell.execute_reply":"2024-11-05T04:58:30.385433Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"#### There seems to be bug in official code, can only submit polars dataframe","metadata":{}},{"cell_type":"code","source":"# 声明一个全局变量 lags_，用于保存滞后数据\n# Declare a global variable lags_ to store lag data\nlags_ : pl.DataFrame | None = None\n\n# 定义预测函数，接收测试集和滞后数据作为输入，返回预测结果\n# Define the prediction function, which takes test data and lag data as inputs, and returns predictions\ndef predict(test: pl.DataFrame, lags: pl.DataFrame | None) -> pl.DataFrame | pd.DataFrame:\n    global lags_  # 使用全局变量 lags_\n    \n    # 如果传入了滞后数据，则将其保存到 lags_\n    # If lag data is provided, save it to lags_\n    if lags is not None:\n        lags_ = lags\n\n    # 创建一个包含 `row_id` 和 `responder_6` 列的预测结果 DataFrame\n    # Create a prediction DataFrame with columns 'row_id' and 'responder_6'\n    predictions = test.select(\n        'row_id',  # 选择 `row_id` 列\n        pl.lit(0.0).alias('responder_6'),  # 将 'responder_6' 列初始化为 0\n    )\n    \n    # 获取所有的 `symbol_id` 列，并将其转换为 NumPy 数组\n    # Extract the 'symbol_id' column and convert it to a NumPy array\n    symbol_ids = test.select('symbol_id').to_numpy()[:, 0]\n\n    # 如果提供了滞后数据，则按 `date_id` 和 `symbol_id` 分组，获取每个组的最后一条记录\n    # If lag data is provided, group by 'date_id' and 'symbol_id' and take the last record of each group\n    if not lags is None:\n        lags = lags.group_by([\"date_id\", \"symbol_id\"], maintain_order=True).last()  # 获取上一日期的最后一条记录\n        test = test.join(lags, on=[\"date_id\", \"symbol_id\"], how=\"left\")  # 将滞后数据与测试数据左连接\n    else:\n        # 如果没有滞后数据，则为每个滞后特征添加默认的 0 值\n        # If no lag data is provided, create columns with default 0 values for lag features\n        test = test.with_columns(\n            (pl.lit(0.0).alias(f'responder_{idx}_lag_1') for idx in range(9))  # 创建 9 个滞后特征，默认值为 0\n        )\n    \n    # 初始化一个全零的预测数组\n    # Initialize a zero array for predictions\n    preds = np.zeros((test.shape[0],))\n    \n    # 使用 XGBoost 模型进行预测，并将结果加到 preds 中\n    # Predict using the XGBoost model and add the result to preds\n    preds += xgb_model.predict(test[xgb_feature_cols].to_pandas()) / 2  # XGBoost 模型的预测结果占 1/2 权重\n\n    # 获取测试数据的特征列，并进行前向填充处理（缺失值填充）\n    # Prepare the test data for neural network models, forward fill missing values\n    test_input = test[CONFIG.feature_cols].to_pandas()\n    test_input = test_input.fillna(method='ffill').fillna(0)  # 填充缺失值，先前向填充，再用 0 填充剩余缺失值\n    test_input = torch.FloatTensor(test_input.values).to(\"cuda:0\")  # 将数据转换为浮动张量并移至 GPU\n\n    # 使用神经网络模型进行预测，累加每个模型的预测结果\n    # Predict using neural network models, accumulate the predictions from each model\n    with torch.no_grad():  # 预测时不计算梯度\n        for i, nn_model in enumerate(tqdm(models)):  # 遍历所有神经网络模型\n            nn_model.eval()  # 设置模型为评估模式\n            preds += nn_model(test_input).cpu().numpy() / 10  # 将每个模型的预测结果累加到 preds 中，权重为 1/10\n\n    # 打印预测结果数组的形状，检查预测结果是否正确\n    # Print the shape of the prediction array to check if the results are correct\n    print(f\"predict> preds.shape =\", preds.shape)\n\n    # 生成最终的预测结果 DataFrame，确保预测值在 -5 到 5 之间\n    # Generate the final prediction DataFrame, ensuring predictions are clipped between -5 and 5\n    predictions = test.select('row_id').\\\n    with_columns(\n        pl.Series(\n            name='responder_6',  # 预测结果列的名称\n            values=np.clip(preds, a_min=-5, a_max=5),  # 将预测结果限制在 -5 到 5 的范围内\n            dtype=pl.Float64,  # 预测列的数据类型为 Float64\n        )\n    )\n\n    # 确保预测函数返回的是 DataFrame\n    # Ensure the prediction function returns a DataFrame\n    assert isinstance(predictions, pl.DataFrame | pd.DataFrame)\n\n    # 确保返回的 DataFrame 包含 'row_id' 和 'responder_6' 两列\n    # Ensure the returned DataFrame has columns 'row_id' and 'responder_6'\n    assert list(predictions.columns) == ['row_id', 'responder_6']\n\n    # 确保预测结果的行数与测试数据相同\n    # Ensure the number of rows in the prediction matches the test data\n    assert len(predictions) == len(test)\n\n    return predictions  # 返回最终的预测结果\n","metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","papermill":{"duration":0.018344,"end_time":"2024-10-10T11:58:33.59684","exception":false,"start_time":"2024-10-10T11:58:33.578496","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2024-11-05T04:58:30.388276Z","iopub.execute_input":"2024-11-05T04:58:30.388681Z","iopub.status.idle":"2024-11-05T04:58:30.401995Z","shell.execute_reply.started":"2024-11-05T04:58:30.388635Z","shell.execute_reply":"2024-11-05T04:58:30.401148Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"When your notebook is run on the hidden test set, inference_server.serve must be called within 15 minutes of the notebook starting or the gateway will throw an error. If you need more than 15 minutes to load your model you can do so during the very first `predict` call, which does not have the usual 10 minute response deadline.","metadata":{"papermill":{"duration":0.002521,"end_time":"2024-10-10T11:58:33.6023","exception":false,"start_time":"2024-10-10T11:58:33.599779","status":"completed"},"tags":[]}},{"cell_type":"code","source":"# 创建一个推理服务器，传入自定义的预测函数\n# Create an inference server and pass the custom prediction function\ninference_server = kaggle_evaluation.jane_street_inference_server.JSInferenceServer(predict)\n\n# 如果是在 Kaggle 竞赛重新运行环境中，则启动推理服务器服务\n# If running in the Kaggle competition rerun environment, start the inference server\nif os.getenv('KAGGLE_IS_COMPETITION_RERUN'):\n    inference_server.serve()  # 启动服务，通常会监听外部请求并返回预测结果\nelse:\n    # 如果不是竞赛重新运行环境，则运行本地网关，使用指定的测试数据和滞后数据进行推理\n    # If not in the competition rerun environment, run a local gateway with the provided test and lag data\n    inference_server.run_local_gateway(\n        (\n            '/kaggle/input/jane-street-realtime-marketdata-forecasting/test.parquet',  # 测试数据文件路径\n            '/kaggle/input/jane-street-realtime-marketdata-forecasting/lags.parquet',   # 滞后数据文件路径\n        )\n    )\n","metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","papermill":{"duration":2.225871,"end_time":"2024-10-10T11:58:35.830964","exception":false,"start_time":"2024-10-10T11:58:33.605093","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2024-11-05T04:58:30.40323Z","iopub.execute_input":"2024-11-05T04:58:30.403628Z","iopub.status.idle":"2024-11-05T04:58:30.495199Z","shell.execute_reply.started":"2024-11-05T04:58:30.403588Z","shell.execute_reply":"2024-11-05T04:58:30.493951Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}