{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":84493,"databundleVersionId":9871156,"sourceType":"competition"},{"sourceId":10120642,"sourceType":"datasetVersion","datasetId":6244857},{"sourceId":10120649,"sourceType":"datasetVersion","datasetId":6244862},{"sourceId":203900450,"sourceType":"kernelVersion"}],"dockerImageVersionId":30805,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# 🚀 Imports and Environment Setup","metadata":{}},{"cell_type":"code","source":"# Data Manipulation\nimport pandas as pd\nimport polars as pl\nimport numpy as np\n\n# OS and Garbage Collection\nimport os\nimport gc\n\n# Progress Bar\nfrom tqdm.auto import tqdm\n\nfrom typing import List\n\n# Visualization Libraries\nfrom matplotlib import pyplot as plt\nimport seaborn as sns\n\n# Serialization\nimport pickle\n\n# Machine Learning Libraries\nfrom sklearn.ensemble import RandomForestRegressor, VotingRegressor\nfrom sklearn.metrics import r2_score\nfrom sklearn.model_selection import train_test_split\nfrom lightgbm import LGBMRegressor\nimport lightgbm as lgb\nfrom xgboost import XGBRegressor\nfrom catboost import CatBoostRegressor\n\n# Deep Learning Libraries\nimport torch\nimport torch.nn as nn\nimport torch.nn.functional as F\nfrom torch.utils.data import Dataset, DataLoader\nfrom pytorch_lightning import (\n    LightningDataModule, \n    LightningModule, \n    Trainer\n)\nfrom pytorch_lightning.callbacks import EarlyStopping, ModelCheckpoint, Timer\n\n# Kaggle-Specific Evaluation Module\nimport kaggle_evaluation.jane_street_inference_server\n\n# Display Options and Warnings\nimport warnings\nwarnings.filterwarnings('ignore')\npd.options.display.max_columns = None\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-21T07:21:52.635105Z","iopub.execute_input":"2024-12-21T07:21:52.635968Z","iopub.status.idle":"2024-12-21T07:21:52.642324Z","shell.execute_reply.started":"2024-12-21T07:21:52.635922Z","shell.execute_reply":"2024-12-21T07:21:52.641478Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# EDA","metadata":{}},{"cell_type":"markdown","source":"# 📂 Data Loading from Parquet Files\nThis cell loads and combines data from Parquet files for efficient processing.\n\n🛠️ **Key Steps**:\n1. **Define the dataset path**: Set the base directory for the data files.\n2. **Iterate through partitions**: Loop over specific partitions of the dataset and load them using pandas.\n3. **Combine datasets**: Concatenate all loaded data into a single DataFrame.\n4. **Round values**: Round all numerical columns to one decimal place.","metadata":{}},{"cell_type":"code","source":"%%time\n\n# Base Dataset Path\npath: str = \"/kaggle/input/jane-street-real-time-market-data-forecasting\"\n\n# List to hold DataFrame samples\nsamples: List[pd.DataFrame] = [] \n\n# Load data from each file partition\nr = range(2)  # Example: Two partitions\nfor i in r:\n    file_path: str = f\"{path}/train.parquet/partition_id={i}/part-0.parquet\"\n    part: pd.DataFrame = pd.read_parquet(file_path)  # Load each partition\n    samples.append(part)  # Append to the samples list\n    \n# Combine all partitions into a single DataFrame\nsample_df: pd.DataFrame = pd.concat(samples, ignore_index=True)\n\n# Round numerical values to one decimal place\nsample_df = sample_df.round(1)\nsample_df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T07:21:52.651965Z","iopub.execute_input":"2024-12-21T07:21:52.652207Z","iopub.status.idle":"2024-12-21T07:21:56.660408Z","shell.execute_reply.started":"2024-12-21T07:21:52.652184Z","shell.execute_reply":"2024-12-21T07:21:56.659410Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 📊 Data Visualization\nThis section visualizes the data with various plots, focusing on:\n\n- **Responder 6 returns and cumulative returns** for symbol_id = 1.\n- **Cumulative response over trade days** for symbol_id = 0.\n- **Correlation heatmap** for responder tags.","metadata":{}},{"cell_type":"code","source":"# Set Grid Color\ngridColor = 'lightgrey'\n\n# DataFrame Setup\ntrain = sample_df\ntrain['N'] = train.index.values  # Assign index values to N column\ntrain['id'] = train.index.values  # Assign index values to id column\n\n# For symbol_id = 1\nxx: pd.Series = sample_df[sample_df.symbol_id == 1]['id']\nyy: pd.Series = sample_df[sample_df.symbol_id == 1]['responder_6']\n\n# Plot Returns for responder_6\nplt.figure(figsize=(16, 5))\nplt.plot(xx, yy, color='black', linewidth=0.05)\nplt.suptitle('Returns, responder_6', weight='bold', fontsize=16)\nplt.xlabel(\"Time\", fontsize=12)\nplt.ylabel(\"Returns\", fontsize=12)\nplt.grid(color=gridColor, linewidth=0.8)\nplt.axhline(0, color='red', linestyle='-', linewidth=1.2)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T07:21:56.662379Z","iopub.execute_input":"2024-12-21T07:21:56.663216Z","iopub.status.idle":"2024-12-21T07:21:57.503533Z","shell.execute_reply.started":"2024-12-21T07:21:56.663169Z","shell.execute_reply":"2024-12-21T07:21:57.502605Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Cumulative responder_6 for symbol_id = 1\nplt.figure(figsize=(14, 4))\nplt.plot(xx, yy.cumsum(), color='black', linewidth=0.6)\nplt.suptitle('Cumulative responder_6', weight='bold', fontsize=16)\nplt.xlabel(\"Time\", fontsize=12)\nplt.ylabel(\"Cumulative res\", fontsize=12)\nplt.yticks(np.arange(-500, 1000, 250))\nplt.grid(color=gridColor)\nplt.axhline(0, color='red', linestyle='-', linewidth=0.7)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T07:21:57.504981Z","iopub.execute_input":"2024-12-21T07:21:57.505801Z","iopub.status.idle":"2024-12-21T07:21:57.783150Z","shell.execute_reply.started":"2024-12-21T07:21:57.505756Z","shell.execute_reply":"2024-12-21T07:21:57.782244Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# For symbol_id = 0, plot cumulative response for all responders\nplt.figure(figsize=(18, 7))\npredictor_cols: List[str] = [col for col in sample_df.columns if 'responder' in col]\nfor i in predictor_cols:\n    if i == 'responder_6': \n        c = 'red'\n        lw = 2.5\n        plt.plot((sample_df[sample_df.symbol_id == 0].groupby(['date_id'])[i].mean()).cumsum(), linewidth=lw, color=c)\n    else: \n        lw = 1\n        plt.plot((sample_df[sample_df.symbol_id == 0].groupby(['date_id'])[i].mean()).cumsum(), linewidth=lw)\n\nplt.xlabel('Trade days')\nplt.ylabel('Cumulative response')\nplt.title('Response time series over trade days  \\n Responder 6 (red) and other responders', weight='bold')\nplt.grid(visible=True, color=gridColor, linewidth=0.7)\nplt.axhline(0, color='blue', linestyle='-', linewidth=1)\nplt.legend(predictor_cols)\nsns.despine()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T07:21:57.785076Z","iopub.execute_input":"2024-12-21T07:21:57.785399Z","iopub.status.idle":"2024-12-21T07:21:59.768643Z","shell.execute_reply.started":"2024-12-21T07:21:57.785367Z","shell.execute_reply":"2024-12-21T07:21:59.767677Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Heatmap of responder correlations from responders.csv\nplt.figure(figsize=(6, 6))\nresponders: pd.DataFrame = pd.read_csv(f\"{path}/responders.csv\")\nmatrix: pd.DataFrame = responders[[f\"tag_{no}\" for no in range(0, 5, 1)]].T.corr()\n\nsns.heatmap(matrix, square=True, cmap=\"coolwarm\", alpha=0.9, vmin=-1, vmax=1, center=0, linewidths=0.5, \n            linecolor='white', annot=True, fmt='.2f')\nplt.xlabel(\"Responder_0 - Responder_8\")\nplt.ylabel(\"Responder_0 - Responder_8\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T07:21:59.769864Z","iopub.execute_input":"2024-12-21T07:21:59.770163Z","iopub.status.idle":"2024-12-21T07:22:00.155120Z","shell.execute_reply.started":"2024-12-21T07:21:59.770135Z","shell.execute_reply":"2024-12-21T07:22:00.154295Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 🧑‍💻 Configuration Class","metadata":{}},{"cell_type":"code","source":"class CONFIG:\n    \"\"\"\n    Configuration class for model setup and data processing.\n\n    This class holds all the necessary configurations, including:\n    - Random seed for reproducibility\n    - Target column name\n    - Feature columns to be used for training\n    - Model paths to the pre-trained models\n\n    Attributes:\n    seed (int): Random seed for reproducibility.\n    target_col (str): The target column to predict in the dataset.\n    feature_cols (List[str]): List of feature columns for model training.\n    model_paths (List[str]): List of paths to pre-trained models.\n    \"\"\"\n\n    seed: int = 42\n    target_col: str = \"responder_6\"\n    \n    # List of feature columns (features and responder lags)\n    feature_cols: List[str] = [\n        f\"feature_{idx:02d}\" for idx in range(79)\n    ] + [\n        f\"responder_{idx}_lag_1\" for idx in range(9)\n    ]\n    \n    # Paths to pre-trained models\n    model_paths: List[str] = [\n        \"/kaggle/input/js-np-nn-trained-model\",\n        \"/kaggle/input/js-lags-trained-xgb/result.pkl\",\n    ]\n\n    def __init__(self):\n        \"\"\"\n        Initializes the CONFIG class with the given configurations.\n        \"\"\"\n        pass\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T07:22:00.156495Z","iopub.execute_input":"2024-12-21T07:22:00.156862Z","iopub.status.idle":"2024-12-21T07:22:00.164093Z","shell.execute_reply.started":"2024-12-21T07:22:00.156823Z","shell.execute_reply":"2024-12-21T07:22:00.163227Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def preprocess_data(data: pd.DataFrame, feature_cols: list) -> torch.FloatTensor:\n    \"\"\"Prepare input data by filling missing values.\"\"\"\n    data = data[feature_cols].fillna(method='ffill').fillna(0)\n    return torch.FloatTensor(data.values)\n\ndef predict_with_models(models, input_data: torch.FloatTensor) -> np.ndarray:\n    \"\"\"Predict using multiple PyTorch models.\"\"\"\n    predictions = np.zeros((input_data.shape[0],))\n    with torch.no_grad():\n        for model in models:\n            model.eval()\n            predictions += model(input_data).cpu().numpy() / len(models)\n    return predictions\n\ndef clear_memory(*variables):\n    \"\"\"Clear variables from memory and force garbage collection.\"\"\"\n    for var in variables:\n        del var\n    gc.collect()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T07:22:00.165091Z","iopub.execute_input":"2024-12-21T07:22:00.165323Z","iopub.status.idle":"2024-12-21T07:22:00.174861Z","shell.execute_reply.started":"2024-12-21T07:22:00.165301Z","shell.execute_reply":"2024-12-21T07:22:00.174153Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 🔮 Prediction","metadata":{}},{"cell_type":"markdown","source":"## 📝 Loading Preprocessed Data for CV Calculation\n","metadata":{}},{"cell_type":"code","source":"def load_preprocessed_data(file_path: str) -> pd.DataFrame:\n    \"\"\"\n    Loads preprocessed data from a Parquet file and converts it into a Pandas DataFrame.\n    \n    Args:\n        file_path (str): Path to the Parquet file containing the preprocessed data.\n    \n    Returns:\n        pd.DataFrame: DataFrame containing the preprocessed data.\n    \n    Raises:\n        FileNotFoundError: If the provided file path does not exist or is incorrect.\n    \"\"\"\n    try:\n        # Load the preprocessed data from the given Parquet file\n        data = pl.scan_parquet(file_path).collect().to_pandas()\n        return data\n    except Exception as e:\n        raise FileNotFoundError(f\"Error loading data from {file_path}: {e}\")\n\n# Define the file path\nfile_path = \"/kaggle/input/js24-preprocessing-create-lags/validation.parquet/\"\n\n# Load the preprocessed validation data\nvalid = load_preprocessed_data(file_path)\nvalid","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T07:22:00.175827Z","iopub.execute_input":"2024-12-21T07:22:00.176162Z","iopub.status.idle":"2024-12-21T07:22:01.095752Z","shell.execute_reply.started":"2024-12-21T07:22:00.176124Z","shell.execute_reply":"2024-12-21T07:22:01.094874Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 📦 Loading the XGB Model","metadata":{}},{"cell_type":"code","source":"xgb_model = None\nmodel_path = CONFIG.model_paths[1]\nwith open( model_path, \"rb\") as fp:\n    result = pickle.load(fp)\n    xgb_model = result[\"model\"]\n\nxgb_feature_cols = [\"symbol_id\", \"time_id\"] + CONFIG.feature_cols\n\n# Show model\ndisplay(xgb_model)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T07:22:01.097042Z","iopub.execute_input":"2024-12-21T07:22:01.097408Z","iopub.status.idle":"2024-12-21T07:22:01.118935Z","shell.execute_reply.started":"2024-12-21T07:22:01.097367Z","shell.execute_reply":"2024-12-21T07:22:01.117933Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 📦 Loading Randomforest model","metadata":{}},{"cell_type":"code","source":"# Initialize and train the Random Forest model\nrf_model = RandomForestRegressor(n_estimators=100, random_state=42)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T07:22:01.125736Z","iopub.execute_input":"2024-12-21T07:22:01.126032Z","iopub.status.idle":"2024-12-21T07:22:01.132017Z","shell.execute_reply.started":"2024-12-21T07:22:01.125997Z","shell.execute_reply":"2024-12-21T07:22:01.128645Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 🧠 Neural Network Model Implementation with Custom R2 Metric and Validation Metrics Logging\r\n","metadata":{}},{"cell_type":"markdown","source":"#### Custom R2 metric and PyTorch Lightning model implementation with validation metrics logging, incorporating sample-weighted R2 calculation, learning rate scheduler, and progress bar visualization.","metadata":{}},{"cell_type":"code","source":"# 📏 Custom R2 Metric Calculation for Validation\ndef r2_val(y_true: np.ndarray, y_pred: np.ndarray, sample_weight: np.ndarray) -> float:\n    \"\"\"\n    Custom R2 metric to calculate weighted R2 score based on sample weights.\n    \n    Args:\n    - y_true (np.ndarray): True target values.\n    - y_pred (np.ndarray): Predicted values by the model.\n    - sample_weight (np.ndarray): Weights for each sample.\n    \n    Returns:\n    - float: Weighted R2 score.\n    \"\"\"\n    r2 = 1 - np.average((y_pred - y_true) ** 2, weights=sample_weight) / (np.average((y_true) ** 2, weights=sample_weight) + 1e-41)\n    return r2\n\n\n# 🧠 Neural Network Model for Regression with PyTorch Lightning\nclass NN(LightningModule):\n    \"\"\"\n    A fully connected neural network model implemented using PyTorch Lightning for regression tasks.\n    The model includes dropout layers, batch normalization, and a custom R2 validation metric.\n    \n    Args:\n    - input_dim (int): The number of input features.\n    - hidden_dims (list): List of integers specifying the hidden layer sizes.\n    - dropouts (list): List of dropout rates for each hidden layer.\n    - lr (float): Learning rate for the optimizer.\n    - weight_decay (float): Weight decay for regularization.\n    \"\"\"\n    \n    def __init__(self, input_dim: int, hidden_dims: list[int], dropouts: list[float], lr: float, weight_decay: float):\n        super().__init__()\n        self.save_hyperparameters()\n        \n        layers = []\n        in_dim = input_dim\n        for i, hidden_dim in enumerate(hidden_dims):\n            layers.append(nn.BatchNorm1d(in_dim))\n            if i > 0:\n                layers.append(nn.SiLU())  # Activation function\n            if i < len(dropouts):\n                layers.append(nn.Dropout(dropouts[i]))\n            layers.append(nn.Linear(in_dim, hidden_dim))\n            in_dim = hidden_dim\n            \n        layers.append(nn.Linear(in_dim, 1))  # Output layer\n        layers.append(nn.Tanh())  # Activation function for output\n        \n        self.model = nn.Sequential(*layers)\n        self.lr = lr\n        self.weight_decay = weight_decay\n        self.validation_step_outputs = []\n\n    def forward(self, x: torch.Tensor) -> torch.Tensor:\n        \"\"\"\n        Forward pass of the model.\n        \n        Args:\n        - x (torch.Tensor): Input tensor.\n        \n        Returns:\n        - torch.Tensor: Model output.\n        \"\"\"\n        return 5 * self.model(x).squeeze(-1)  # Output as 1D tensor\n\n    def training_step(self, batch: tuple[torch.Tensor, torch.Tensor, torch.Tensor]) -> torch.Tensor:\n        \"\"\"\n        Training step for the model.\n        \n        Args:\n        - batch (tuple[torch.Tensor, torch.Tensor, torch.Tensor]): A tuple of input data (x), targets (y), and sample weights (w).\n        \n        Returns:\n        - torch.Tensor: The computed loss for the current batch.\n        \"\"\"\n        x, y, w = batch\n        y_hat = self(x)\n        loss = F.mse_loss(y_hat, y, reduction='none') * w  # Weighted MSE loss\n        loss = loss.mean()  # Mean of the weighted losses\n        self.log('train_loss', loss, on_step=False, on_epoch=True, batch_size=x.size(0))\n        return loss\n\n    def validation_step(self, batch: tuple[torch.Tensor, torch.Tensor, torch.Tensor]) -> torch.Tensor:\n        \"\"\"\n        Validation step for the model.\n        \n        Args:\n        - batch (tuple[torch.Tensor, torch.Tensor, torch.Tensor]): A tuple of input data (x), targets (y), and sample weights (w).\n        \n        Returns:\n        - torch.Tensor: The computed loss for the current validation batch.\n        \"\"\"\n        x, y, w = batch\n        y_hat = self(x)\n        loss = F.mse_loss(y_hat, y, reduction='none') * w\n        loss = loss.mean()\n        self.log('val_loss', loss, on_step=False, on_epoch=True, batch_size=x.size(0))\n        self.validation_step_outputs.append((y_hat, y, w))\n        return loss\n\n    def on_validation_epoch_end(self):\n        \"\"\"\n        Calculate and log validation R2 score at the end of each validation epoch.\n        \"\"\"\n        y = torch.cat([x[1] for x in self.validation_step_outputs]).cpu().numpy()\n        prob = torch.cat([x[0] for x in self.validation_step_outputs]).cpu().numpy()\n        weights = torch.cat([x[2] for x in self.validation_step_outputs]).cpu().numpy()\n        \n        # Calculate weighted R2 score\n        val_r_square = r2_val(y, prob, weights)\n        self.log(\"val_r_square\", val_r_square, prog_bar=True, on_step=False, on_epoch=True)\n        self.validation_step_outputs.clear()\n\n    def configure_optimizers(self) -> dict:\n        \"\"\"\n        Configure the optimizer and learning rate scheduler for the model.\n        \n        Returns:\n        - dict: Dictionary containing optimizer and learning rate scheduler configuration.\n        \"\"\"\n        optimizer = torch.optim.Adam(self.parameters(), lr=self.lr, weight_decay=self.weight_decay)\n        scheduler = torch.optim.lr_scheduler.ReduceLROnPlateau(optimizer, mode='min', factor=0.5, patience=5, verbose=True)\n        return {\n            'optimizer': optimizer,\n            'lr_scheduler': {\n                'scheduler': scheduler,\n                'monitor': 'val_loss',\n            }\n        }\n\n    def on_train_epoch_end(self):\n        \"\"\"\n        Log metrics at the end of each training epoch.\n        \"\"\"\n        if self.trainer.sanity_checking:\n            return\n        epoch = self.trainer.current_epoch\n        metrics = {k: v.item() if isinstance(v, torch.Tensor) else v for k, v in self.trainer.logged_metrics.items()}\n        formatted_metrics = {k: f\"{v:.5f}\" for k, v in metrics.items()}\n        print(f\"Epoch {epoch}: {formatted_metrics}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T07:22:01.133442Z","iopub.execute_input":"2024-12-21T07:22:01.134008Z","iopub.status.idle":"2024-12-21T07:22:01.162374Z","shell.execute_reply.started":"2024-12-21T07:22:01.133968Z","shell.execute_reply":"2024-12-21T07:22:01.161393Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"#### 🔥 Load and initialize the best models for `N_folds` cross-validation using PyTorch Lightning checkpoints on GPU.\r\n","metadata":{}},{"cell_type":"code","source":"N_folds = 5\n#  loading the best models\nmodels = []\nfor fold in range(N_folds):\n    checkpoint_path = f\"{CONFIG.model_paths[0]}/nn_{fold}.model\"\n    model = NN.load_from_checkpoint(checkpoint_path)\n    models.append(model.to(\"cuda:0\"))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T07:22:01.163339Z","iopub.execute_input":"2024-12-21T07:22:01.163596Z","iopub.status.idle":"2024-12-21T07:22:01.384877Z","shell.execute_reply.started":"2024-12-21T07:22:01.163566Z","shell.execute_reply":"2024-12-21T07:22:01.384209Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 📊 Calculate Cross-Validation (CV) Score using XGBoost and Sample Weights\r\n","metadata":{}},{"cell_type":"code","source":"X_valid = valid[ xgb_feature_cols ]\ny_valid = valid[ CONFIG.target_col ]\nw_valid = valid[ \"weight\" ]\ny_pred_valid_xgb = xgb_model.predict(X_valid)\nvalid_score = r2_score( y_valid, y_pred_valid_xgb, sample_weight=w_valid )\nvalid_score","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T07:22:01.385860Z","iopub.execute_input":"2024-12-21T07:22:01.386203Z","iopub.status.idle":"2024-12-21T07:22:03.975752Z","shell.execute_reply.started":"2024-12-21T07:22:01.386167Z","shell.execute_reply":"2024-12-21T07:22:03.974677Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_valid = preprocess_data(valid, CONFIG.feature_cols)\ny_valid = valid[CONFIG.target_col].values\nw_valid = valid[\"weight\"].values\n\nprint(\"Validation data prepared:\", X_valid.shape, y_valid.shape, w_valid.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T07:22:03.977131Z","iopub.execute_input":"2024-12-21T07:22:03.977404Z","iopub.status.idle":"2024-12-21T07:22:07.054653Z","shell.execute_reply.started":"2024-12-21T07:22:03.977378Z","shell.execute_reply":"2024-12-21T07:22:07.053732Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y_pred_valid_nn = predict_with_models(models, X_valid.to(\"cuda:0\"))\nvalid_score = r2_score(y_valid, y_pred_valid_nn, sample_weight=w_valid)\nprint(f\"Validation R2 score (NN): {valid_score:.5f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T07:22:07.055792Z","iopub.execute_input":"2024-12-21T07:22:07.056099Z","iopub.status.idle":"2024-12-21T07:22:08.922365Z","shell.execute_reply.started":"2024-12-21T07:22:07.056070Z","shell.execute_reply":"2024-12-21T07:22:08.921222Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"#### Prepare validation data by filling missing values in features with forward fill and zeros, then output shapes of `X_valid`, `y_valid`, and `w_valid`.\n","metadata":{}},{"cell_type":"code","source":"X_valid = valid[ CONFIG.feature_cols ]\ny_valid = valid[ CONFIG.target_col ]\nw_valid = valid[ \"weight\" ]\nX_valid = X_valid.fillna(method = 'ffill').fillna(0)\nX_valid.shape, y_valid.shape, w_valid.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T07:22:08.923762Z","iopub.execute_input":"2024-12-21T07:22:08.924483Z","iopub.status.idle":"2024-12-21T07:22:11.076883Z","shell.execute_reply.started":"2024-12-21T07:22:08.924429Z","shell.execute_reply":"2024-12-21T07:22:11.076052Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# y_pred_valid_nn = predict_with_models(models, X_valid.to(\"cuda:0\"))\n# valid_score = r2_score(y_valid, y_pred_valid_nn, sample_weight=w_valid)\n# print(f\"Validation R2 score (NN): {valid_score:.5f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T07:22:11.077990Z","iopub.execute_input":"2024-12-21T07:22:11.078264Z","iopub.status.idle":"2024-12-21T07:22:11.081975Z","shell.execute_reply.started":"2024-12-21T07:22:11.078237Z","shell.execute_reply":"2024-12-21T07:22:11.080998Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 🤖 Train and Evaluate ElasticNet Model with R2 Score\r\n","metadata":{}},{"cell_type":"code","source":"from sklearn.linear_model import ElasticNet\nfrom sklearn.metrics import r2_score\n\n# Train ElasticNet model\nen_model = ElasticNet(alpha=0.1, l1_ratio=1)  # You can adjust alpha and l1_ratio as needed\nen_model.fit(X_valid, y_valid, sample_weight=w_valid)\n\n# Predict with the trained ElasticNet model\ny_pred_valid_en = en_model.predict(X_valid)\n\n# Evaluate the model's performance using R2 score\nvalid_score_en = r2_score(y_valid, y_pred_valid_en, sample_weight=w_valid)\nprint(f\"Validation R2 score for ElasticNet: {valid_score_en}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T07:22:11.083270Z","iopub.execute_input":"2024-12-21T07:22:11.083554Z","iopub.status.idle":"2024-12-21T07:22:17.260475Z","shell.execute_reply.started":"2024-12-21T07:22:11.083525Z","shell.execute_reply":"2024-12-21T07:22:17.256906Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Train Gradient Boosting Model (LightGBM)\nlgb_model = LGBMRegressor(n_estimators=1000, learning_rate=0.02, max_depth=7)\nlgb_model.fit(X_valid, y_valid, sample_weight=w_valid)\n\n# Predict with the trained LightGBM model\ny_pred_valid_lgb = lgb_model.predict(X_valid)\nvalid_score_lgb = r2_score(y_valid, y_pred_valid_lgb, sample_weight=w_valid)\nprint(f\"Validation R2 score for LightGBM: {valid_score_lgb}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T07:22:17.262117Z","iopub.execute_input":"2024-12-21T07:22:17.262728Z","iopub.status.idle":"2024-12-21T07:25:31.401519Z","shell.execute_reply.started":"2024-12-21T07:22:17.262661Z","shell.execute_reply":"2024-12-21T07:25:31.400571Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"#### Aggregate predictions from all models for validation data using PyTorch, calculate the weighted R2 score, and store the validation score.","metadata":{}},{"cell_type":"code","source":"y_pred_valid_nn = np.zeros(y_valid.shape)\nwith torch.no_grad():\n    for model in models:\n        model.eval()\n        y_pred_valid_nn += model(torch.FloatTensor(X_valid.values).to(\"cuda:0\")).cpu().numpy() / len(models)\nvalid_score = r2_score( y_valid, y_pred_valid_nn, sample_weight=w_valid )\nvalid_score","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T07:25:31.402883Z","iopub.execute_input":"2024-12-21T07:25:31.403292Z","iopub.status.idle":"2024-12-21T07:25:38.213752Z","shell.execute_reply.started":"2024-12-21T07:25:31.403251Z","shell.execute_reply":"2024-12-21T07:25:38.212820Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"#### Combine predictions from XGBoost and neural network models in a weighted ensemble, compute the weighted R2 score, and store the validation score.","metadata":{}},{"cell_type":"code","source":"y_pred_valid_ensemble = 0.5 * (y_pred_valid_xgb + y_pred_valid_nn)\nvalid_score = r2_score( y_valid, y_pred_valid_ensemble, sample_weight=w_valid )\nvalid_score","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T07:25:38.215205Z","iopub.execute_input":"2024-12-21T07:25:38.216296Z","iopub.status.idle":"2024-12-21T07:25:38.234296Z","shell.execute_reply.started":"2024-12-21T07:25:38.216251Z","shell.execute_reply":"2024-12-21T07:25:38.233363Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"#### Release validation data from memory and trigger garbage collection to optimize memory usage.","metadata":{}},{"cell_type":"code","source":"del valid, X_valid, y_valid, w_valid\ngc.collect()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T07:25:38.235610Z","iopub.execute_input":"2024-12-21T07:25:38.236002Z","iopub.status.idle":"2024-12-21T07:25:38.515786Z","shell.execute_reply.started":"2024-12-21T07:25:38.235970Z","shell.execute_reply":"2024-12-21T07:25:38.514876Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"#### Define a prediction function that processes test data, integrates lag features, combines predictions from XGBoost, neural network, and LightGBM models, and outputs clipped predictions as a DataFrame.\n","metadata":{}},{"cell_type":"code","source":"lags_ : pl.DataFrame | None = None\ndef predict(test: pl.DataFrame, lags: pl.DataFrame | None) -> pl.DataFrame | pd.DataFrame:\n    global lags_\n    if lags is not None:\n        lags_ = lags\n\n    predictions = test.select(\n        'row_id',\n        pl.lit(0.0).alias('responder_6'),\n    )\n    \n    symbol_ids = test.select('symbol_id').to_numpy()[:, 0]\n\n    if lags is not None:\n        lags = lags.group_by([\"date_id\", \"symbol_id\"], maintain_order=True).last() # pick up last record of previous date\n        test = test.join(lags, on=[\"date_id\", \"symbol_id\"], how=\"left\")\n    else:\n        test = test.with_columns(\n            (pl.lit(0.0).alias(f'responder_{idx}_lag_1') for idx in range(9))\n        )\n\n    preds = np.zeros((test.shape[0],))\n    \n    # Adding XGBoost predictions\n    preds += xgb_model.predict(test[xgb_feature_cols].to_pandas()) / 3\n    \n    # Adding Neural Network predictions\n    test_input = test[CONFIG.feature_cols].to_pandas()\n    test_input = test_input.fillna(method='ffill').fillna(0)\n    test_input = torch.FloatTensor(test_input.values).to(\"cuda:0\")\n    with torch.no_grad():\n        for nn_model in models:\n            nn_model.eval()\n            preds += nn_model(test_input).cpu().numpy() / len(models)\n    \n    # Adding LightGBM predictions\n    preds += lgb_model.predict(test[CONFIG.feature_cols].to_pandas()) / 3\n    \n    print(f\"predict> preds.shape =\", preds.shape)\n    \n    predictions = test.select('row_id').with_columns(\n        pl.Series(\n            name='responder_6',\n            values=np.clip(preds, a_min=-5, a_max=5),\n            dtype=pl.Float64,\n        )\n    )\n\n    assert isinstance(predictions, pl.DataFrame | pd.DataFrame)\n    assert list(predictions.columns) == ['row_id', 'responder_6']\n    assert len(predictions) == len(test)\n\n    return predictions\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T07:25:38.517096Z","iopub.execute_input":"2024-12-21T07:25:38.517467Z","iopub.status.idle":"2024-12-21T07:25:38.531388Z","shell.execute_reply.started":"2024-12-21T07:25:38.517426Z","shell.execute_reply":"2024-12-21T07:25:38.530743Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 🚀 Inference Server Setup for Submission: Local and Remote Execution with Real-Time Data 📊\r\n","metadata":{}},{"cell_type":"code","source":"inference_server = kaggle_evaluation.jane_street_inference_server.JSInferenceServer(predict)\n\nif os.getenv('KAGGLE_IS_COMPETITION_RERUN'):\n    inference_server.serve()\nelse:\n    inference_server.run_local_gateway(\n        (\n            '/kaggle/input/jane-street-real-time-market-data-forecasting/test.parquet',\n            '/kaggle/input/jane-street-real-time-market-data-forecasting/lags.parquet',\n        )\n    )","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T07:25:38.532347Z","iopub.execute_input":"2024-12-21T07:25:38.532578Z","iopub.status.idle":"2024-12-21T07:25:39.078340Z","shell.execute_reply.started":"2024-12-21T07:25:38.532554Z","shell.execute_reply":"2024-12-21T07:25:39.077425Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}