{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":84493,"databundleVersionId":9871156,"sourceType":"competition"},{"sourceId":9801075,"sourceType":"datasetVersion","datasetId":6006872},{"sourceId":9806342,"sourceType":"datasetVersion","datasetId":6010899},{"sourceId":203900450,"sourceType":"kernelVersion"},{"sourceId":206712546,"sourceType":"kernelVersion"},{"sourceId":207025929,"sourceType":"kernelVersion"},{"sourceId":207107641,"sourceType":"kernelVersion"},{"sourceId":207107951,"sourceType":"kernelVersion"}],"dockerImageVersionId":30786,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport polars as pl\nimport numpy as np\nimport os, gc\nfrom tqdm.auto import tqdm\nfrom matplotlib import pyplot as plt\nimport pickle\n\nimport torch\nimport torch.nn as nn\nimport torch.nn.functional as F\nfrom pytorch_lightning import (LightningDataModule, LightningModule, Trainer)\nfrom pytorch_lightning.callbacks import EarlyStopping, ModelCheckpoint, Timer\n\nimport pandas as pd\nimport numpy as np\nfrom sklearn.metrics import r2_score\nfrom sklearn.model_selection import train_test_split\nfrom torch.utils.data import Dataset, DataLoader\n\n\nfrom sklearn.metrics import r2_score\nfrom lightgbm import LGBMRegressor\nimport lightgbm as lgb\nfrom xgboost import XGBRegressor\nfrom catboost import CatBoostRegressor\nfrom sklearn.ensemble import VotingRegressor\n\nimport warnings\nwarnings.filterwarnings('ignore')\npd.options.display.max_columns = None\n\nimport kaggle_evaluation.jane_street_inference_server","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-11-16T21:13:25.192683Z","iopub.execute_input":"2024-11-16T21:13:25.193581Z","iopub.status.idle":"2024-11-16T21:13:25.202207Z","shell.execute_reply.started":"2024-11-16T21:13:25.193538Z","shell.execute_reply":"2024-11-16T21:13:25.201051Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<span style=\"font-size:24px; font-weight:bold\">Configuration</span>","metadata":{}},{"cell_type":"markdown","source":"This code defines a CONFIG class that sets up configuration parameters for a machine learning model. It initializes the random seed (seed = 42) to ensure reproducibility of results. The target_col is set to \"responder_6\", indicating the target variable for prediction. It defines feature_cols as a list of 79 feature columns (feature_xx) and 9 lag-1 responder columns (responder_xx_lag_1). Additionally, it provides a list of model_paths, which includes paths to pre-trained models, including a neural network model and an XGBoost model, which will be used for prediction or further training.","metadata":{}},{"cell_type":"code","source":"class CONFIG:\n\n    # Set random seed for getting the same results\n\n    seed = 42\n\n    # The target column (the variable that the model will predict)\n\n    target_col = \"responder_6\"\n\n    # Feature columns: consist of 79 \"feature_xx\" columns and 9 lag-1 \"resonder_xx_lag_1\" columns\n\n    feature_cols = [f\"feature_{idx:02d}\" for idx in range(79)] + [f\"responder_{idx}_lag_1\" for idx in range(9)]\n\n    # Model paths: list of paths to pretrained models\n\n    model_paths = [\n        #\"/kaggle/input/js24-train-gbdt-model-with-lags-singlemodel/result.pkl\",\n        #\"/kaggle/input/js24-trained-gbdt-model/result.pkl\",\n        \"/kaggle/input/js-xs-nn-trained-model\", # Pre-trained NN Model\n        \"/kaggle/input/js-with-lags-trained-xgb/result.pkl\" #Path to pre-trained XGBoost model\n        \n    ]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-16T21:13:25.204079Z","iopub.execute_input":"2024-11-16T21:13:25.204491Z","iopub.status.idle":"2024-11-16T21:13:25.215799Z","shell.execute_reply.started":"2024-11-16T21:13:25.204458Z","shell.execute_reply":"2024-11-16T21:13:25.214893Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<span style=\"font-size:24px; font-weight:bold\">Load preprocessed data (to calculate CV)</span>","metadata":{}},{"cell_type":"markdown","source":"This code reads a Parquet file (validation.parquet) from a specified path, loads it into a Polars DataFrame, collects the data, and converts it into a Pandas DataFrame for further use.","metadata":{}},{"cell_type":"code","source":"valid = pl.scan_parquet(\n    f\"/kaggle/input/js24-preprocessing-create-lags/validation.parquet/\"\n).collect().to_pandas()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-16T21:13:25.217078Z","iopub.execute_input":"2024-11-16T21:13:25.217478Z","iopub.status.idle":"2024-11-16T21:13:26.034179Z","shell.execute_reply.started":"2024-11-16T21:13:25.217434Z","shell.execute_reply":"2024-11-16T21:13:26.033144Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<span style=\"font-size:24px; font-weight:bold\">Local model</span>","metadata":{}},{"cell_type":"markdown","source":"This code begins by initializing the xgb_model variable to None, indicating that the model has not yet been loaded. It then selects the second path from the CONFIG.model_paths list, which points to a pre-trained XGBoost model. The model file is opened in binary read mode (\"rb\"), and the contents are deserialized using pickle.load(). The deserialized content is expected to be a dictionary, from which the XGBoost model is extracted under the key \"model\" and assigned to the xgb_model variable. Next, the xgb_feature_cols list is defined, which includes the columns \"symbol_id\", \"time_id\", and the feature columns from CONFIG.feature_cols. Finally, the loaded XGBoost model is displayed using the display() function. This process allows the code to load a pre-trained model and prepare the necessary feature set for making predictions or evaluating the model.","metadata":{}},{"cell_type":"code","source":"\n# Initialize the XGB_MODEL variable to None\n\nxgb_model = None\n\n# Get the model path from CONFIG configuration, select the second XGBoost model path\n\nmodel_paths = CONFIG.model_paths[1]\n\n# Open the model file and load the model\n\nwith open(model_paths, \"rb\") as fp:\n\n    # Use pickle to load the contents of the model file\n\n    result = pickle.load(fp)\n\n    # Extract the model object from the loaded result and store it in \"xgb_model\"\n\n    xgb_model = result[\"model\"]\n\n# Define the feature columns for the model, including \"symbol_id\", \"time_id\" and the feature columns from the CONFIG configuration\n\nxgb_feature_cols = [\"symbol_id\", \"time_id\"] + CONFIG.feature_cols\n\n# Display the loaded XGBoost model\n\ndisplay(xgb_model)\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-16T21:13:26.035768Z","iopub.execute_input":"2024-11-16T21:13:26.036073Z","iopub.status.idle":"2024-11-16T21:13:26.061173Z","shell.execute_reply.started":"2024-11-16T21:13:26.036041Z","shell.execute_reply":"2024-11-16T21:13:26.059989Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<span style=\"font-size:24px; font-weight:bold\">Custom R2 metric for evaluation</span>","metadata":{}},{"cell_type":"markdown","source":"\nThis code defines a neural network model class NN that extends LightningModule from PyTorch Lightning. The model is designed for a regression task with weighted mean squared error loss (MSE).\n\nThe r2_val function calculates a weighted R² score, which is a measure of how well the model's predictions match the actual data, considering the sample weights. The function avoids division by zero by adding a small constant (1e-38) to the denominator.\n\nThe NN class itself implements several key functions for training, validation, and optimization. In the __init__ method, the network architecture is defined with batch normalization, activation functions (SiLU), dropout layers, and fully connected layers. The final output layer uses the Tanh activation to normalize outputs.\n\nDuring training and validation (training and validation methods), the model computes weighted MSE loss using sample weights. The model's performance is tracked with logged loss values. In the validation step, at the end of each epoch, the on_validation_epoch_end method calculates the R² score, logging it as val_r_square.\n\nThe configure_optimizers method configures the Adam optimizer and a learning rate scheduler (ReduceLROnPlateau) to adjust the learning rate based on validation loss. Finally, on_train_epoch_end prints out training metrics after each epoch.","metadata":{}},{"cell_type":"code","source":"def r2_val(y_true, y_pred, sample_weight):\n\n    # Calculate weighted R2 value, with  1e-38 to avoid division from zero\n\n    r2 = 1 - np.average((y_pred - y_true) ** 2, weights=sample_weight) / (np.average((y_true) ** 2, weights = sample_weight) + 1e-38)\n    return r2\n\n\nclass NN(LightningModule):\n\n    # Initialize the neural network model\n\n    def __init__(self, input_dim, hidden_dims, dropouts, lr, weight_decay):\n        super().__init__()\n\n        # Save hyperparameters for use in training\n\n        self.save_hyperparameters\n\n        layers = []\n        in_dim = input_dim\n\n        # Construct the hidden layers, adding BatchNorm, activation function, Dropout, and Linear layers step by step\n\n        for i, hidden_dim in enumerate(hidden_dims):\n\n            # Adding a Batch Normalization layer\n            \n            layers.append(nn.BatchNorm1d(in_dim))\n\n            # Activation function SiLu (Sigmoid Linear Unit)\n            \n            if i > 0:\n                layers.append(nn.SiLU())\n\n            # Dropout layer to avoid overfitting\n            \n            if i < len(dropouts):\n                layers.append(nn.Dropout(dropouts[i]))\n            \n            # Fully connected layers (linear transformation)\n            \n            layers.append(nn.Linear(in_dim, hidden_dim)) \n\n            # Update input dimenstion\n            \n            in_dim = hidden_dim\n\n        # Output layer, the last one is linear\n        \n        layers.append(nn.Linear(in_dim, 1))\n\n        # Use the Tanh activation function to normalize the output values\n        \n        layers.append(nn.Tanh())\n\n        # Compose all layers into a Sequential model\n        \n        self.model = nn.Sequential(*layers)\n\n        # Learning rate\n        \n        self.lr = lr\n\n        # Weight decay, L2 regularization\n\n        self.weight_decay = weight_decay \n\n        # Used to save the output of every validation step\n\n        self.validation_step_outputs = []\n\n    # Forward pass function\n    \n    def forward(self, x):\n\n        # Output passes through the model, scaled by 5 and removes extra dimensions\n        \n        return 5 * self.model(x).squeeze(-1)  \n\n    # Training step\n    \n    def training(self, batch):\n\n        # Get input data x, label y and weight w\n        \n        x, y, w = batch\n\n        # Predictions using model\n        \n        y_hat = self(x)\n\n        # Calculate weighted MSE losses\n\n        loss = F.mse_loss(y_hat, reduction='none') * w\n\n        # Average losses across all samples\n\n        loss = loss.mean()\n\n        # Record training loss\n\n        self.log('train_loss', loss, on_step=False, on_epoch=True, batch_size=x.size(0))\n\n        # Returns loss value\n\n        \n\n    # Validation step\n    \n    def validation(self, batch):\n\n        # Get input data x, label y and weight w\n        \n        x, y, w = batch\n\n        # Predictions using model\n        \n        y_hat = self(x)\n\n        # Calculate weighted MSE losses\n\n        loss = F.mse_loss(y_hat, reduction='none') * w\n\n        # Average losses across all samples\n\n        loss = loss.mean()\n\n        # Record validation loss\n\n        self.log('val_loss', loss, on_step=False, on_epoch=True, batch_size=x.size(0))\n\n        # Save the output for each step for the use in calculating other metrics\n\n        self.validation_step_outputs.append((y_hat, y, w))\n\n        # Returns loss value\n\n        return loss\n\n\n\n\n    # Called at the end of each validation epoch to calculate R2 on the validation \n    def on_validation_epoch_end(self):\n        \"\"\"Calculate validation WRMSE at the end of the epoch.\"\"\"\n\n        # Get all labelled values\n\n        y = torch.cat([x[1] for x in self.validation_step_outputs]).cpu().numpy()\n\n        if self.trainer.sanity_checking:\n\n            # If its a validation step, get the prediction directly\n\n            prob = torch.cat([x[0] for x in self.validation_step_outputs]).cpu().numpy()\n\n        else:\n\n            # Get all predicted values\n\n            prob = torch.cat([x[0] for x in self.validation_step_outputs]).cpu().numpy()\n\n            # Get all the weights\n\n            weights = torch.cat([x[2] for x in self.validation_step_outputs]).cpu().numpy()\n\n            # Calculate the weighted R2 value\n\n            val_r_square = r2_val(y, prob, weights)\n\n            # Record the R2 values for the validation step\n\n            self.log(\"val_r_square\", val_r_square, prog_bar=True, on_step=True, on_epoch=True)\n\n        # Clear the output step for the next cycle\n\n        self.validation_step_outputs.clear()\n\n\n    # Configure the optimizer and learning rate scheduler\n\n    def configure_optimizers(self):\n\n        # Use the ReduceLROnPlateau Learning Rate Scheduler to reduce learning rates when verification losses dont decline\n        \n        optimizer = torch.optim.Adam(self.parameters(), lr=self.lr, weight_decay=self.weight_decay)\n\n        scheduler = torch.optim.lr_scheduler.ReduceLROnPlateau(optimizer, mode='min', factor=0.5, patience=5, verbose = True)\n\n        return {'optimizer':optimizer,  # Return optimizer\n                'lr_scheduler' :{\n                    'scheduler' : scheduler,  # Return learning rate scheduler\n                    'monitor' : 'val_loss'    # Monitor validation loss to adjust learning rate\n                }\n               }\n\n     # Called at the end of each training epoch to print metrics\n\n    def on_train_epoch_end(self):\n\n        if self.trainer.sanity_checking:\n\n            return  # If its a validation step, skip the output\n\n        epoch = self.trainer.current_epoch  # Current training epoch\n\n        # Get all the metrics from the training process\n\n        metrics = {k: v.items() if isinstance(v, torch.Tensor) else v for k, v in self.trainer.logged_metrics.items()}\n\n        # Format output metrics to 5 decimal places\n\n        formatted_metrics = {k: f\"{v:.5f}\" for k, v in metrics.items()}\n\n        # Print metrics for the current training cycle\n\n        print(f\"Epoch {epoch}: {formatted_metrics}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-16T21:13:26.065538Z","iopub.execute_input":"2024-11-16T21:13:26.066077Z","iopub.status.idle":"2024-11-16T21:13:26.112981Z","shell.execute_reply.started":"2024-11-16T21:13:26.066023Z","shell.execute_reply":"2024-11-16T21:13:26.111839Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"This code performs cross-validation by loading the best model from each fold in a 5-fold setup. It iterates through each fold, constructs the checkpoint path, loads the model using load_from_checkpoint(), and moves the model to the GPU (CUDA device 0). The models are stored in a list for later use.","metadata":{}},{"cell_type":"code","source":"# Set the number of folds for cross-validation\n\nN_folds = 5\n\n# Initialize an emply list to store the trained models\n\nmodels = []\n\n# Loop through each fold and load the best model\n\nfor fold in range(N_folds):\n\n    # Construct the checkpoint path for each fold model\n\n    checkpoint_path = f\"{CONFIG.model_paths[0]}/nn_{fold}.model\"\n\n    # Load the model from the checkpoint file\n\n    model = NN.load_from_checkpoint(checkpoint_path)\n\n    # Move the model to the GPU (assumed to be CUDA device 0)\n\n    models.append(model.to(\"cuda:0\"))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-16T21:13:26.114636Z","iopub.execute_input":"2024-11-16T21:13:26.115121Z","iopub.status.idle":"2024-11-16T21:13:26.425902Z","shell.execute_reply.started":"2024-11-16T21:13:26.115051Z","shell.execute_reply":"2024-11-16T21:13:26.424868Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"This code extracts the feature columns (X_valid) and target variable (y_valid) from the validation set. It also retrieves sample weights (w_valid). The XGBoost model is used to make predictions (y_pred_valid_xgb) on the validation set. Finally, it calculates and outputs the weighted R² score (valid_score) using the true labels, predictions, and sample weights.","metadata":{}},{"cell_type":"code","source":"# Extract feature columns X_valid and target variable y_valid from the validation set\n\nX_valid = valid[xgb_feature_cols] # Extract feature columns\ny_valid = valid[CONFIG.target_col] # Extract target columns(labels)\n\n# Extract sample weights from the validation set\n\nw_valid = valid[\"weight\"]\n\n# Predict the validation set using the XGBoost model\n\ny_pred_valid_xgb = xgb_model.predict(X_valid)\n\n# Calculate the weighted R2 score (validation R2 score)\n\nvalid_score = r2_score(y_valid, y_pred_valid_xgb, sample_weight=w_valid)\n\n# Output the validation R2 score\n\nvalid_score","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-16T21:13:26.427168Z","iopub.execute_input":"2024-11-16T21:13:26.427468Z","iopub.status.idle":"2024-11-16T21:13:29.446357Z","shell.execute_reply.started":"2024-11-16T21:13:26.427435Z","shell.execute_reply":"2024-11-16T21:13:29.445057Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"This code extracts the feature columns (X_valid) and target variable (y_valid) from the validation set, along with sample weights (w_valid). It then fills any missing values in X_valid using forward fill (ffill) and replaces any remaining missing values with 0. Finally, it outputs the shapes of X_valid, y_valid, and w_valid.","metadata":{}},{"cell_type":"code","source":"# Extract feature columns X_valid and target variable y_valid from the validation set\n\nX_valid = valid[CONFIG.feature_cols]  # Extract feature columns (using the feature_cols list in the configuration)\ny_valid = valid[CONFIG.target_col]  # Extract target columns (labels)\n\n# Extract sample weights from the validation set\n\nw_valid = valid[\"weight\"]\n\n# Fill missing values in X_valid:\n# 1. Forward fill ('ffill') to fill missing values.\n# 2. If still missing, fill with 0.\n\nX_valid = X_valid.fillna(method='ffill').fillna(0)\n\n# Output the shapes of X_valid, y_valid, and w_valid\n\nX_valid.shape, y_valid.shape, w_valid.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-16T21:13:29.447694Z","iopub.execute_input":"2024-11-16T21:13:29.448091Z","iopub.status.idle":"2024-11-16T21:13:31.584243Z","shell.execute_reply.started":"2024-11-16T21:13:29.448052Z","shell.execute_reply":"2024-11-16T21:13:31.583247Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"This code initializes an array (y_pred_valid_nn) to store predictions from multiple neural network models. It then iterates over the models, setting each to evaluation mode and making predictions in inference mode (without gradients). The validation features are converted to a FloatTensor, moved to the GPU, and predictions are accumulated, averaging across all models. Finally, it calculates and outputs the weighted R² score (valid_score) using the true labels, predicted values, and sample weights.","metadata":{}},{"cell_type":"code","source":"# Initialize an array of zeros, y_pred_valid_nn, to store predictions from the neural network models\n\ny_pred_valid_nn = np.zeros(y_valid.shape)\n\n# Make predictions without calculating gradients (inference mode)\n\nwith torch.no_grad():\n    for model in models:\n        model.eval()  # Set the model to evaluation mode (disable Dropout and BatchNorm)\n     \n        # Convert validation features to FloatTensor and move them to GPU\n        \n        y_pred_valid_nn += model(torch.FloatTensor(X_valid.values).to(\"cuda:0\")).cpu().numpy() / len(models)\n      \n        # Take the average of predictions from all models\n\n\n# Calculate the weighted R2 score (validation R2 score)\n\nvalid_score = r2_score(y_valid, y_pred_valid_nn, sample_weight=w_valid)\n\n# Output the validation R2 score\n\nvalid_score","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-16T21:13:31.585454Z","iopub.execute_input":"2024-11-16T21:13:31.585775Z","iopub.status.idle":"2024-11-16T21:13:38.204094Z","shell.execute_reply.started":"2024-11-16T21:13:31.585741Z","shell.execute_reply":"2024-11-16T21:13:38.203154Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"\nThis code combines predictions from both the XGBoost and neural network models using a weighted average, where each model contributes equally (50%). It then calculates the weighted R² score for the ensemble model using the true labels, the combined predictions, and the sample weights. Finally, it outputs the R² score (valid_score) for the ensemble model.","metadata":{}},{"cell_type":"code","source":"\n# Combine XGBoost and neural network model predictions with weighted average\n\ny_pred_valid_ensemble = 0.5 * (y_pred_valid_xgb + y_pred_valid_nn)\n\n# Calculate the weighted R2 score (R2 score for the ensemble model)\n\nvalid_score = r2_score(y_valid, y_pred_valid_ensemble, sample_weight=w_valid)\n\n# Output the R2 score for the ensemble model\n\nvalid_score","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-16T21:13:38.205673Z","iopub.execute_input":"2024-11-16T21:13:38.206531Z","iopub.status.idle":"2024-11-16T21:13:38.226932Z","shell.execute_reply.started":"2024-11-16T21:13:38.206479Z","shell.execute_reply":"2024-11-16T21:13:38.225868Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"This code deletes the variables valid, X_valid, y_valid, and w_valid to free up memory, as they are no longer needed. It then explicitly triggers garbage collection using gc.collect() to ensure that the memory used by these variables is released and returned to the system, helping optimize memory usage during the program's execution.","metadata":{}},{"cell_type":"code","source":"\n# Delete variables that are no longer needed to free up memory\n\ndel valid, X_valid, y_valid, w_valid\n\n# Explicitly trigger garbage collection to release unused memory\n\ngc.collect()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-16T21:13:38.229079Z","iopub.execute_input":"2024-11-16T21:13:38.229764Z","iopub.status.idle":"2024-11-16T21:13:38.462729Z","shell.execute_reply.started":"2024-11-16T21:13:38.229712Z","shell.execute_reply":"2024-11-16T21:13:38.461632Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"This code defines a predict function that generates predictions based on a provided test dataset and optional lag data, returning the results as a DataFrame.\n\nA global variable lags_ is declared to store lag data when provided. If lag data (lags) is supplied, it is saved to lags_ and used to join with the test data, adding relevant historical features. If no lag data is provided, default lag columns with 0 values are added to the test set.\n\nThe function then prepares predictions using two models: an XGBoost model and a set of neural network models. For XGBoost, predictions are weighted by 0.5 and added to an initial zeroed prediction array. For the neural network models, the test data is forward-filled for missing values, converted to a tensor, and predictions are accumulated from each model, with each model’s prediction weighted by 0.1.\n\nThe final predictions are clipped between -5 and 5 and returned in a DataFrame with row_id and responder_6 columns. Several assertions ensure the function's integrity, confirming that the returned DataFrame has the correct structure, dimensions, and column names. The function outputs the combined and weighted predictions from both models.","metadata":{}},{"cell_type":"code","source":"\n# Declare a global variable lags_ to store lag data\n\nlags_ : pl.DataFrame | None = None\n\n\n# Define the prediction function, which takes test data and lag data as inputs, and returns predictions\n\ndef predict(test: pl.DataFrame, lags: pl.DataFrame | None) -> pl.DataFrame | pd.DataFrame:\n    global lags_  # Using the global variable lags_\n    \n\n    # If lag data is provided, save it to lags_\n    \n    if lags is not None:\n        lags_ = lags\n\n\n    # Create a prediction DataFrame with columns 'row_id' and 'responder_6'\n    \n    predictions = test.select(\n        'row_id',  # Select the `row_id` column\n        pl.lit(0.0).alias('responder_6'),  # Initialize 'responder_6' column to 0\n    )\n    \n    # Extract the 'symbol_id' column and convert it to a NumPy array\n    \n    symbol_ids = test.select('symbol_id').to_numpy()[:, 0]\n\n    # If lag data is provided, group by 'date_id' and 'symbol_id' and take the last record of each group\n    \n    if not lags is None:\n        lags = lags.group_by([\"date_id\", \"symbol_id\"], maintain_order=True).last()  # Get the last record of the previous date\n        test = test.join(lags, on=[\"date_id\", \"symbol_id\"], how=\"left\")  # Left-connect hysteresis data to test data\n    else:\n        \n        # If no lag data is provided, create columns with default 0 values for lag features\n        \n        test = test.with_columns(\n            (pl.lit(0.0).alias(f'responder_{idx}_lag_1') for idx in range(9))  # Create 9 lag features, default value is 0\n        )\n    \n\n    # Initialize a zero array for predictions\n    \n    preds = np.zeros((test.shape[0],))\n    \n    # Predict using the XGBoost model and add the result to preds\n    \n    preds += xgb_model.predict(test[xgb_feature_cols].to_pandas()) / 2  # 1/2 weight on XGBoost model predictions\n\n    # Prepare the test data for neural network models, forward fill missing values\n    \n    test_input = test[CONFIG.feature_cols].to_pandas()\n    test_input = test_input.fillna(method='ffill').fillna(0)  # Fill in missing values, pre-populate, then fill in the remaining missing values with zeros\n    test_input = torch.FloatTensor(test_input.values).to(\"cuda:0\")  # Convert data to a floating tensor and move to the GPU\n\n\n    # Predict using neural network models, accumulate the predictions from each model\n    \n    with torch.no_grad():  # Gradients are not calculated for prediction\n        for i, nn_model in enumerate(tqdm(models)):  # Iterate over all neural network models\n            nn_model.eval()  # Setting up the model for evaluation mode\n            preds += nn_model(test_input).cpu().numpy() / 10  # Accumulate the predictions of each model into preds with a weight of 1/10\n\n    # Print the shape of the prediction array to check if the results are correct\n    \n    print(f\"predict> preds.shape =\", preds.shape)\n\n    # Generate the final prediction DataFrame, ensuring predictions are clipped between -5 and 5\n    \n    predictions = test.select('row_id').\\\n    with_columns(\n        pl.Series(\n            name='responder_6',  # Name of the forecast result column\n            values=np.clip(preds, a_min=-5, a_max=5),  # Limit predictions to a range of -5 to 5\n            dtype=pl.Float64,  # The data type of the predicted column is Float64\n        )\n    )\n\n  \n    # Ensure the prediction function returns a DataFrame\n    \n    assert isinstance(predictions, pl.DataFrame | pd.DataFrame)\n\n    # Ensure the returned DataFrame has columns 'row_id' and 'responder_6'\n    \n    assert list(predictions.columns) == ['row_id', 'responder_6']\n\n    # Ensure the number of rows in the prediction matches the test data\n    \n    assert len(predictions) == len(test)\n\n    return predictions  # Returns the final prediction","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-16T21:13:38.464491Z","iopub.execute_input":"2024-11-16T21:13:38.464855Z","iopub.status.idle":"2024-11-16T21:13:38.480166Z","shell.execute_reply.started":"2024-11-16T21:13:38.464818Z","shell.execute_reply":"2024-11-16T21:13:38.479316Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"When your notebook is run on the hidden test set, inference_server.serve must be called within 15 minutes of the notebook starting or the gateway will throw an error. If you need more than 15 minutes to load your model you can do so during the very first predict call, which does not have the usual 10 minute response deadline.","metadata":{}},{"cell_type":"code","source":"inference_server = kaggle_evaluation.jane_street_inference_server.JSInferenceServer(predict)\n\nif os.getenv('KAGGLE_IS_COMPETITION_RERUN'):\n    inference_server.serve()\nelse:\n    inference_server.run_local_gateway(\n        (\n            '/kaggle/input/jane-street-real-time-market-data-forecasting/test.parquet',\n            '/kaggle/input/jane-street-real-time-market-data-forecasting/lags.parquet',\n        )\n    )\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-16T21:13:38.481498Z","iopub.execute_input":"2024-11-16T21:13:38.481894Z","iopub.status.idle":"2024-11-16T21:13:38.580741Z","shell.execute_reply.started":"2024-11-16T21:13:38.481848Z","shell.execute_reply":"2024-11-16T21:13:38.579758Z"}},"outputs":[],"execution_count":null}]}