{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"codemirror_mode":{"name":"ipython","version":3},"file_extension":".py","mimetype":"text/x-python","name":"python","nbconvert_exporter":"python","pygments_lexer":"ipython3","version":"3.10.14"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":84493,"databundleVersionId":9871156,"sourceType":"competition"},{"sourceId":9801075,"sourceType":"datasetVersion","datasetId":6006872},{"sourceId":9806342,"sourceType":"datasetVersion","datasetId":6010899},{"sourceId":10139918,"sourceType":"datasetVersion","datasetId":6258261},{"sourceId":10139922,"sourceType":"datasetVersion","datasetId":6258265},{"sourceId":10253875,"sourceType":"datasetVersion","datasetId":6297065},{"sourceId":10304887,"sourceType":"datasetVersion","datasetId":6378806},{"sourceId":10351700,"sourceType":"datasetVersion","datasetId":6410107},{"sourceId":10452425,"sourceType":"datasetVersion","datasetId":6470258},{"sourceId":203900450,"sourceType":"kernelVersion"},{"sourceId":215616115,"sourceType":"kernelVersion"},{"sourceId":216017958,"sourceType":"kernelVersion"},{"sourceId":216577393,"sourceType":"kernelVersion"}],"dockerImageVersionId":30787,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true},"papermill":{"default_parameters":{},"duration":31.696539,"end_time":"2025-01-13T14:46:14.123326","environment_variables":{},"exception":null,"input_path":"__notebook__.ipynb","output_path":"__notebook__.ipynb","parameters":{},"start_time":"2025-01-13T14:45:42.426787","version":"2.6.0"},"widgets":{"application/vnd.jupyter.widget-state+json":{"state":{"252dd2de87de42f4becd877fbbafc26b":{"model_module":"@jupyter-widgets/controls","model_module_version":"1.5.0","model_name":"HTMLModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"1.5.0","_model_name":"HTMLModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"1.5.0","_view_name":"HTMLView","description":"","description_tooltip":null,"layout":"IPY_MODEL_ff55584ae0ab4f39ae471f314eaad988","placeholder":"​","style":"IPY_MODEL_b8d338c473aa4dc3ba25f37a997a9037","value":" 1/1 [00:00&lt;00:00, 33.79it/s]"}},"48a2731fb59b4ce8ace5b53d6f0e3337":{"model_module":"@jupyter-widgets/controls","model_module_version":"1.5.0","model_name":"HTMLModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"1.5.0","_model_name":"HTMLModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"1.5.0","_view_name":"HTMLView","description":"","description_tooltip":null,"layout":"IPY_MODEL_a8dbc79c7c5a48318466a102c55801bb","placeholder":"​","style":"IPY_MODEL_ebd06aaaa7024a1684ba1b9fe89358cf","value":"100%"}},"56da652f3eeb42aca986e6fd815629dc":{"model_module":"@jupyter-widgets/base","model_module_version":"1.2.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"1.2.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"1.2.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"overflow_x":null,"overflow_y":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"5bb4e0df01c44716afebab25aafe9f5f":{"model_module":"@jupyter-widgets/controls","model_module_version":"1.5.0","model_name":"FloatProgressModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"1.5.0","_model_name":"FloatProgressModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"1.5.0","_view_name":"ProgressView","bar_style":"success","description":"","description_tooltip":null,"layout":"IPY_MODEL_f4e41234b0cc4f3e9313d971f5aadc9f","max":1,"min":0,"orientation":"horizontal","style":"IPY_MODEL_6408698650f74dd699bcc914b850e396","value":1}},"6408698650f74dd699bcc914b850e396":{"model_module":"@jupyter-widgets/controls","model_module_version":"1.5.0","model_name":"ProgressStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"1.5.0","_model_name":"ProgressStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"1.2.0","_view_name":"StyleView","bar_color":null,"description_width":""}},"9cb493ec04fc4ed391f5ac28cc84500e":{"model_module":"@jupyter-widgets/controls","model_module_version":"1.5.0","model_name":"HBoxModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"1.5.0","_model_name":"HBoxModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"1.5.0","_view_name":"HBoxView","box_style":"","children":["IPY_MODEL_48a2731fb59b4ce8ace5b53d6f0e3337","IPY_MODEL_5bb4e0df01c44716afebab25aafe9f5f","IPY_MODEL_252dd2de87de42f4becd877fbbafc26b"],"layout":"IPY_MODEL_56da652f3eeb42aca986e6fd815629dc"}},"a8dbc79c7c5a48318466a102c55801bb":{"model_module":"@jupyter-widgets/base","model_module_version":"1.2.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"1.2.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"1.2.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"overflow_x":null,"overflow_y":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"b8d338c473aa4dc3ba25f37a997a9037":{"model_module":"@jupyter-widgets/controls","model_module_version":"1.5.0","model_name":"DescriptionStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"1.5.0","_model_name":"DescriptionStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"1.2.0","_view_name":"StyleView","description_width":""}},"ebd06aaaa7024a1684ba1b9fe89358cf":{"model_module":"@jupyter-widgets/controls","model_module_version":"1.5.0","model_name":"DescriptionStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"1.5.0","_model_name":"DescriptionStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"1.2.0","_view_name":"StyleView","description_width":""}},"f4e41234b0cc4f3e9313d971f5aadc9f":{"model_module":"@jupyter-widgets/base","model_module_version":"1.2.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"1.2.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"1.2.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"overflow_x":null,"overflow_y":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"ff55584ae0ab4f39ae471f314eaad988":{"model_module":"@jupyter-widgets/base","model_module_version":"1.2.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"1.2.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"1.2.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"overflow_x":null,"overflow_y":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}}},"version_major":2,"version_minor":0}}},"nbformat_minor":4,"nbformat":4,"cells":[{"id":"85143825","cell_type":"markdown","source":"# Additional\n\n- 2024/12/24 : find weights ensemble and add in predict() preds = (w) * preds_xgb + (w-1) * preds_nn\n- 2024/12/26 : XGB local train modelx5 add. dataset is here\n    - https://www.kaggle.com/code/hideyukizushi/js-nnx5-xgbx6-weighted-blend-lb-0-0077\n- 2025/01/02 : add weights ensemble of public notebook\n    - https://www.kaggle.com/code/yunsuxiaozi/js-ridge-baseline?scriptVersionId=202739388\n    - https://www.kaggle.com/code/hideyukizushi/js-nnx5-xgbx5-weighted-blend-lb-0-0078/notebook?scriptVersionId=215419804\n    - https://www.kaggle.com/code/i2nfinit3y/jane-street-tabm-ft-transformer-inference?scriptVersionId=213715783","metadata":{"papermill":{"duration":0.007146,"end_time":"2025-01-13T14:45:44.980645","exception":false,"start_time":"2025-01-13T14:45:44.973499","status":"completed"},"tags":[]}},{"id":"6c1a8b4a","cell_type":"code","source":"!pip install rtdl_num_embeddings -q --no-index --find-links=/kaggle/input/jane-street-import/rtdl_num_embeddings","metadata":{"execution":{"iopub.status.busy":"2025-01-13T20:05:32.215210Z","iopub.execute_input":"2025-01-13T20:05:32.215648Z","iopub.status.idle":"2025-01-13T20:05:42.583979Z","shell.execute_reply.started":"2025-01-13T20:05:32.215613Z","shell.execute_reply":"2025-01-13T20:05:42.582427Z"},"papermill":{"duration":9.454654,"end_time":"2025-01-13T14:45:54.441689","exception":false,"start_time":"2025-01-13T14:45:44.987035","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"id":"7660314c","cell_type":"code","source":"import os, sys, gc\nimport pickle\nimport dill\nimport numpy as np\nimport pandas as pd\nimport polars as pl\n\nimport torch\nimport torch.nn as nn\nimport torch.nn.functional as F\nfrom pytorch_lightning import (LightningDataModule, LightningModule, Trainer)\n\nfrom sklearn.metrics import r2_score\n\nimport torch.optim\nfrom torch.utils.data import Dataset, DataLoader, TensorDataset\nfrom sklearn.model_selection import train_test_split\nimport math\nfrom tqdm import tqdm\nfrom collections import OrderedDict\nfrom tabm_reference import Model, make_parameter_groups\n\nimport warnings\nimport joblib\nfrom pytorch_lightning.callbacks import Callback\nimport gc\n\nimport lightgbm as lgb\nfrom lightgbm import LGBMRegressor, Booster\nfrom xgboost import XGBRegressor\nfrom catboost import CatBoostRegressor\n\nimport warnings\nwarnings.filterwarnings('ignore')\npd.options.display.max_columns = None\n\nsys.path.append(\"/kaggle/input/jane-street-real-time-market-data-forecasting\")","metadata":{"execution":{"iopub.status.busy":"2025-01-13T20:05:42.586630Z","iopub.execute_input":"2025-01-13T20:05:42.587085Z","iopub.status.idle":"2025-01-13T20:05:42.596732Z","shell.execute_reply.started":"2025-01-13T20:05:42.587029Z","shell.execute_reply":"2025-01-13T20:05:42.595475Z"},"papermill":{"duration":11.400963,"end_time":"2025-01-13T14:46:05.849086","exception":false,"start_time":"2025-01-13T14:45:54.448123","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"id":"cc65956c","cell_type":"code","source":"folder11 = \"/kaggle/input/jane-street-real-time-market-data-forecasting/test.parquet/date_id=0/part-0.parquet\"\nf = pl.read_parquet(folder11)\nf.head()","metadata":{"execution":{"iopub.status.busy":"2025-01-13T20:05:42.598243Z","iopub.execute_input":"2025-01-13T20:05:42.598603Z","iopub.status.idle":"2025-01-13T20:05:42.623750Z","shell.execute_reply.started":"2025-01-13T20:05:42.598567Z","shell.execute_reply":"2025-01-13T20:05:42.622630Z"},"papermill":{"duration":0.152431,"end_time":"2025-01-13T14:46:06.013068","exception":false,"start_time":"2025-01-13T14:46:05.860637","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"id":"b29a58d8","cell_type":"markdown","source":"# Top Public Notebook","metadata":{"papermill":{"duration":0.006764,"end_time":"2025-01-13T14:46:06.026429","exception":false,"start_time":"2025-01-13T14:46:06.019665","status":"completed"},"tags":[]}},{"id":"ddbf2799","cell_type":"markdown","source":"## JS|NNx5+XGBx5(MyTrain+Pub)|WeightBlend|LB.0.0078 (hideyukizushi)\n\n`LB: 0.0078` https://www.kaggle.com/code/yunsuxiaozi/js2024-starter?scriptVersionId=206770572","metadata":{"papermill":{"duration":0.006119,"end_time":"2025-01-13T14:46:06.039146","exception":false,"start_time":"2025-01-13T14:46:06.033027","status":"completed"},"tags":[]}},{"id":"539e87eb","cell_type":"code","source":"class CONFIG:\n    \"\"\"Configuration class for model parameters\"\"\"\n    seed = 42  # Random seed for reproducibility\n    target_col = \"responder_6\"  # Target variable name\n    # Features: 79 base features + 9 lagged features\n    feature_cols = [f\"feature_{idx:02d}\" for idx in range(79)] + [f\"responder_{idx}_lag_1\" for idx in range(9)]\n    # Paths to pre-trained models\n    model_paths = [\n        \"/kaggle/input/js-xs-nn-trained-model\",  # Neural Network models\n        \"/kaggle/input/js-with-lags-trained-xgb/result.pkl\", # XGBoost model\n        \"/kaggle/input/als-e-106-pp0-40-xgb-5fold/result0.pkl\",\n        \"/kaggle/input/als-e-106-pp0-40-xgb-5fold/result1.pkl\",\n        \"/kaggle/input/als-e-106-pp0-40-xgb-5fold/result2.pkl\",\n        \"/kaggle/input/als-e-106-pp0-40-xgb-5fold/result3.pkl\",\n        \"/kaggle/input/als-e-106-pp0-40-xgb-5fold/result4.pkl\"  \n    ]\n\n# Load validation data\nvalid = pl.scan_parquet(f\"/kaggle/input/js24-preprocessing-create-lags/validation.parquet/\").collect().to_pandas()\n\n# Load XGBoost model\nxgb_model = None\nwith open(CONFIG.model_paths[2], \"rb\") as fp:\n    result = pickle.load(fp)\n    xgb_model = result[\"model\"]\nxgb_feature_cols = [\"symbol_id\", \"time_id\"] + CONFIG.feature_cols\n\nxgb_model2 = None\nwith open(CONFIG.model_paths[4], \"rb\") as fp:\n    result = pickle.load(fp)\n    xgb_model2 = result[\"model\"]\n\nxgb_model4 = None\nwith open(CONFIG.model_paths[6], \"rb\") as fp:\n    result = pickle.load(fp)\n    xgb_model4 = result[\"model\"]\n\nxgb_model5 = None\nwith open(CONFIG.model_paths[1], \"rb\") as fp:\n    result = pickle.load(fp)\n    xgb_model5 = result[\"model\"]\n\ndef r2_val(y_true, y_pred, sample_weight):\n    \"\"\"\n    Calculate weighted R² score\n    Args:\n        y_true: True values\n        y_pred: Predicted values\n        sample_weight: Weights for each sample\n    Returns:\n        Weighted R² score\n    \"\"\"\n    r2 = 1 - np.average((y_pred - y_true) ** 2, weights=sample_weight) / (np.average((y_true) ** 2, weights=sample_weight) + 1e-38)\n    return r2\n\nclass NN(LightningModule):\n    \"\"\"Neural Network model using PyTorch Lightning\"\"\"\n    \n    def __init__(self, input_dim, hidden_dims, dropouts, lr, weight_decay):\n        \"\"\"\n        Initialize the neural network\n        Args:\n            input_dim: Input feature dimension\n            hidden_dims: List of hidden layer dimensions\n            dropouts: List of dropout rates\n            lr: Learning rate\n            weight_decay: Weight decay for regularization\n        \"\"\"\n        super().__init__()\n        self.save_hyperparameters()\n        \n        # Build network architecture\n        layers = []\n        in_dim = input_dim\n        for i, hidden_dim in enumerate(hidden_dims):\n            layers.append(nn.BatchNorm1d(in_dim))  # Batch normalization\n            if i > 0:\n                layers.append(nn.SiLU())  # SiLU activation (except first layer)\n            if i < len(dropouts):\n                layers.append(nn.Dropout(dropouts[i]))  # Dropout for regularization\n            layers.append(nn.Linear(in_dim, hidden_dim))  # Linear layer\n            in_dim = hidden_dim\n            \n        # Output layer\n        layers.append(nn.Linear(in_dim, 1))\n        layers.append(nn.Tanh())  # Tanh activation for bounded output\n        \n        self.model = nn.Sequential(*layers)\n        self.lr = lr\n        self.weight_decay = weight_decay\n        self.validation_step_outputs = []\n\n    def forward(self, x):\n        \"\"\"Forward pass with scaling\"\"\"\n        return 5 * self.model(x).squeeze(-1)  # Scale output to [-5, 5] range\n\n    def training_step(self, batch):\n        \"\"\"Single training step\"\"\"\n        x, y, w = batch\n        y_hat = self(x)\n        loss = F.mse_loss(y_hat, y, reduction='none') * w  # Weighted MSE loss\n        loss = loss.mean()\n        self.log('train_loss', loss, on_step=False, on_epoch=True, batch_size=x.size(0))\n        return loss\n\n    def validation_step(self, batch):\n        \"\"\"Single validation step\"\"\"\n        x, y, w = batch\n        y_hat = self(x)\n        loss = F.mse_loss(y_hat, y, reduction='none') * w\n        loss = loss.mean()\n        self.log('val_loss', loss, on_step=False, on_epoch=True, batch_size=x.size(0))\n        self.validation_step_outputs.append((y_hat, y, w))\n        return loss\n\n    def on_validation_epoch_end(self):\n        \"\"\"Compute validation metrics at epoch end\"\"\"\n        if not self.trainer.sanity_checking:\n            y = torch.cat([x[1] for x in self.validation_step_outputs]).cpu().numpy()\n            prob = torch.cat([x[0] for x in self.validation_step_outputs]).cpu().numpy()\n            weights = torch.cat([x[2] for x in self.validation_step_outputs]).cpu().numpy()\n            val_r_square = r2_val(y, prob, weights)\n            self.log(\"val_r_square\", val_r_square, prog_bar=True, on_step=False, on_epoch=True)\n        self.validation_step_outputs.clear()\n\n    def configure_optimizers(self):\n        \"\"\"Configure optimizer and learning rate scheduler\"\"\"\n        optimizer = torch.optim.Adam(self.parameters(), lr=self.lr, weight_decay=self.weight_decay)\n        scheduler = torch.optim.lr_scheduler.ReduceLROnPlateau(\n            optimizer, \n            mode='min', \n            factor=0.5, \n            patience=5, \n            verbose=True\n        )\n        return {\n            'optimizer': optimizer,\n            'lr_scheduler': {\n                'scheduler': scheduler,\n                'monitor': 'val_loss',\n            }\n        }\n\n    def on_train_epoch_end(self):\n        \"\"\"Log metrics at end of training epoch\"\"\"\n        if not self.trainer.sanity_checking:\n            epoch = self.trainer.current_epoch\n            metrics = {k: v.item() if isinstance(v, torch.Tensor) else v \n                      for k, v in self.trainer.logged_metrics.items()}\n            formatted_metrics = {k: f\"{v:.5f}\" for k, v in metrics.items()}\n            print(f\"Epoch {epoch}: {formatted_metrics}\")\n\n# Load ensemble of models (5-fold cross-validation)\nN_folds = 5\nmodels = []\nfor fold in range(N_folds):\n    checkpoint_path = f\"{CONFIG.model_paths[0]}/nn_{fold}.model\"\n    model = NN.load_from_checkpoint(checkpoint_path)\n    models.append(model.to(\"cuda:0\"))","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2025-01-13T20:05:42.626546Z","iopub.execute_input":"2025-01-13T20:05:42.626917Z","iopub.status.idle":"2025-01-13T20:05:44.274324Z","shell.execute_reply.started":"2025-01-13T20:05:42.626882Z","shell.execute_reply":"2025-01-13T20:05:44.273385Z"},"papermill":{"duration":3.334998,"end_time":"2025-01-13T14:46:09.380177","exception":false,"start_time":"2025-01-13T14:46:06.045179","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"id":"2c97d638","cell_type":"code","source":"# Clear validation data from memory to free up space\n#del valid\ngc.collect()\n\n# Global variable to store lagged features\nlags_: pl.DataFrame | None = None\n\ndef predict_nn_xgb(test: pl.DataFrame, lags: pl.DataFrame | None) -> pl.DataFrame | pd.DataFrame:\n    \"\"\"\n    Make predictions using ensemble of XGBoost and Neural Network models\n    \n    Args:\n        test: DataFrame containing test data\n        lags: DataFrame containing lagged features (optional)\n        \n    Returns:\n        DataFrame with predictions\n    \"\"\"\n    global lags_\n    \n    # Store lags in global variable if provided\n    if lags is not None:\n        lags_ = lags\n\n    # Initialize predictions DataFrame with row_id and placeholder predictions\n    predictions_nn = test.select('row_id', pl.lit(0.0).alias('responder_6',))\n\n    # Process lagged features\n    # Get last record for each date_id and symbol_id combination\n    lags = lags_.clone().group_by([\"date_id\", \"symbol_id\"], maintain_order=True).last()\n    \n    # Join test data with lagged features\n    test = test.join(lags, on=[\"date_id\", \"symbol_id\"], how=\"left\")\n\n    # Initialize arrays for model predictions\n    preds_xgb = np.zeros((test.shape[0],))  # XGBoost predictions\n    preds_nn = np.zeros((test.shape[0],))   # Neural Network predictions\n\n    # Generate XGBoost predictions\n    preds_xgb += xgb_model.predict(test[xgb_feature_cols].to_pandas()) * 0.25\n    preds_xgb += xgb_model2.predict(test[xgb_feature_cols].to_pandas()) * 0.25\n    preds_xgb += xgb_model4.predict(test[xgb_feature_cols].to_pandas()) * 0.25\n    preds_xgb += xgb_model5.predict(test[xgb_feature_cols].to_pandas()) * 0.25\n\n    # Generate Neural Network predictions\n    # Prepare input data\n    test_input = test[CONFIG.feature_cols].to_pandas()\n    # Handle missing values: forward fill then fill remaining with zeros\n    test_input = test_input.fillna(method='ffill').fillna(0)\n    # Convert to PyTorch tensor and move to GPU\n    test_input = torch.FloatTensor(test_input.values).to(\"cuda:0\")\n\n    # Generate predictions from Neural Network ensemble\n    with torch.no_grad():  # Disable gradient calculation for inference\n        for i, nn_model in enumerate(models):\n            nn_model.eval()  # Set model to evaluation mode\n            # Average predictions from all models\n            preds_nn += nn_model(test_input).cpu().numpy() / len(models)\n\n    # Combine predictions with equal weights (50% XGBoost, 50% Neural Network)\n    preds = 0.55 * preds_xgb + 0.45 * preds_nn\n\n    # Create final predictions DataFrame\n    predictions_nn = test.select('row_id').\\\n        with_columns(\n            pl.Series(\n                name='responder_6',\n                values=np.clip(preds, a_min=-5, a_max=5),  # Clip predictions to [-5, 5] range\n                dtype=pl.Float64,\n            )\n        )\n\n    return predictions_nn","metadata":{"_kg_hide-input":false,"_kg_hide-output":false,"execution":{"iopub.status.busy":"2025-01-13T20:05:44.275661Z","iopub.execute_input":"2025-01-13T20:05:44.275986Z","iopub.status.idle":"2025-01-13T20:05:44.588299Z","shell.execute_reply.started":"2025-01-13T20:05:44.275955Z","shell.execute_reply":"2025-01-13T20:05:44.587235Z"},"papermill":{"duration":0.234899,"end_time":"2025-01-13T14:46:09.621560","exception":false,"start_time":"2025-01-13T14:46:09.386661","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"id":"c8eb7d2a","cell_type":"markdown","source":"## Jane Street | TabM/FT-Transformer inference (i2nfinit3y)\n`LB: 0.0074` https://www.kaggle.com/code/i2nfinit3y/jane-street-tabm-ft-transformer-inference?scriptVersionId=213715783","metadata":{"papermill":{"duration":0.005874,"end_time":"2025-01-13T14:46:09.633930","exception":false,"start_time":"2025-01-13T14:46:09.628056","status":"completed"},"tags":[]}},{"id":"c2d59bfe","cell_type":"code","source":"feature_list = [f\"feature_{idx:02d}\" for idx in range(79) if idx != 61]\n\ntarget_col = \"responder_6\" \n\nfeature_test = feature_list \\\n                + [f\"responder_{idx}_lag_1\" for idx in range(9)] \n\nfeature_cat = [\"feature_09\", \"feature_10\", \"feature_11\"]\nfeature_cont = [item for item in feature_test if item not in feature_cat]\n\nbatch_size = 8192\n\nstd_feature = [i for i in feature_list if i not in feature_cat] + [f\"responder_{idx}_lag_1\" for idx in range(9)]\n\ndata_stats = joblib.load(\"/kaggle/input/my-own-js/data_stats.pkl\")\nmeans = data_stats['mean']\nstds = data_stats['std']\n\ndef standardize(df, feature_cols, means, stds):\n    return df.with_columns([\n        ((pl.col(col) - means[col]) / stds[col]).alias(col) for col in feature_cols\n    ])\n\ncategory_mappings = {'feature_09': {2: 0, 4: 1, 9: 2, 11: 3, 12: 4, 14: 5, 15: 6, 25: 7, 26: 8, 30: 9, 34: 10, 42: 11, 44: 12, 46: 13, 49: 14, 50: 15, 57: 16, 64: 17, 68: 18, 70: 19, 81: 20, 82: 21},\n 'feature_10': {1: 0, 2: 1, 3: 2, 4: 3, 5: 4, 6: 5, 7: 6, 10: 7, 12: 8},\n 'feature_11': {9: 0, 11: 1, 13: 2, 16: 3, 24: 4, 25: 5, 34: 6, 40: 7, 48: 8, 50: 9, 59: 10, 62: 11, 63: 12, 66: 13,\n  76: 14, 150: 15, 158: 16, 159: 17, 171: 18, 195: 19, 214: 20, 230: 21, 261: 22, 297: 23, 336: 24, 376: 25, 388: 26, 410: 27, 522: 28, 534: 29, 539: 30},\n 'symbol_id': {0: 0, 1: 1, 2: 2, 3: 3, 4: 4, 5: 5, 6: 6, 7: 7, 8: 8, 9: 9, 10: 10, 11: 11, 12: 12, 13: 13, 14: 14, 15: 15, 16: 16, 17: 17, 18: 18, 19: 19,\n  20: 20, 21: 21, 22: 22, 23: 23, 24: 24, 25: 25, 26: 26, 27: 27, 28: 28, 29: 29, 30: 30, 31: 31, 32: 32, 33: 33, 34: 34, 35: 35, 36: 36, 37: 37, 38: 38},\n 'time_id' : {i : i for i in range(968)}}\n\ndef encode_column(df, column, mapping):\n    max_value = max(mapping.values())  \n\n    def encode_category(category):\n        return mapping.get(category, max_value + 1)  \n    \n    return df.with_columns(\n        pl.col(column).map_elements(encode_category).alias(column)\n    )\n\nclass R2Loss(nn.Module):\n    def __init__(self):\n        super(R2Loss, self).__init__()\n\n    def forward(self, y_pred, y_true):\n        mse_loss = torch.sum((y_pred - y_true) ** 2)\n        var_y = torch.sum(y_true ** 2)\n        loss = mse_loss / (var_y + 1e-38)\n        return loss\n\nclass NN(LightningModule):\n    def __init__(self, n_cont_features, cat_cardinalities, n_classes, lr, weight_decay):\n        super().__init__()\n        self.save_hyperparameters()\n        self.k = 16\n        self.model = Model(\n                n_num_features=n_cont_features,\n                cat_cardinalities=cat_cardinalities,\n                n_classes=n_classes,\n                backbone={\n                    'type': 'MLP',\n                    'n_blocks': 3 ,\n                    'd_block': 512,\n                    'dropout': 0.25,\n                },\n                bins=None,\n                num_embeddings= None,\n                arch_type='tabm',\n                k=self.k,\n            )\n        self.lr = lr\n        self.weight_decay = weight_decay\n        self.training_step_outputs = []\n        self.validation_step_outputs = []\n        self.loss_fn = R2Loss()\n        # self.loss_fn = weighted_mse_loss\n\n    def forward(self, x_cont, x_cat):\n        return self.model(x_cont, x_cat).squeeze(-1)\n\n    def training_step(self, batch):\n        x_cont,x_cat, y, w , w_y= batch\n        x_cont = x_cont + torch.randn_like(x_cont) * 0.02\n        y_hat = self(x_cont, x_cat)\n        # loss = self.loss_fn(y_hat.flatten(0, 1), y.repeat_interleave(self.k), w_y.repeat_interleave(self.k))\n        loss = self.loss_fn(y_hat.flatten(0, 1), y.repeat_interleave(self.k))\n        self.log('train_loss', loss, on_step=True, on_epoch=True, prog_bar=True, logger=True, batch_size=x_cont.size(0))\n        self.training_step_outputs.append((y_hat.mean(1), y, w))\n        return loss\n\n    def validation_step(self, batch):\n        x_cont,x_cat, y, w, w_y = batch\n        x_cont = x_cont + torch.randn_like(x_cont) * 0.02\n        y_hat = self(x_cont, x_cat)\n        # loss = self.loss_fn(y_hat.flatten(0, 1), y.repeat_interleave(self.k), w_y.repeat_interleave(self.k))\n        loss = self.loss_fn(y_hat.flatten(0, 1), y.repeat_interleave(self.k))\n        self.log('val_loss', loss, on_step=False, on_epoch=True, prog_bar=True, logger=True, batch_size=x_cont.size(0))\n        self.validation_step_outputs.append((y_hat.mean(1), y, w))\n        return loss\n\n    def on_validation_epoch_end(self):\n        \"\"\"Calculate validation WRMSE at the end of the epoch.\"\"\"\n        y = torch.cat([x[1] for x in self.validation_step_outputs]).cpu().numpy()\n        if self.trainer.sanity_checking:\n            prob = torch.cat([x[0] for x in self.validation_step_outputs]).cpu().numpy()\n        else:\n            prob = torch.cat([x[0] for x in self.validation_step_outputs]).cpu().numpy()\n            weights = torch.cat([x[2] for x in self.validation_step_outputs]).cpu().numpy()\n            # r2_val\n            val_r_square = r2_val(y, prob, weights)\n            self.log(\"val_r_square\", val_r_square, prog_bar=True, on_step=False, on_epoch=True)\n        self.validation_step_outputs.clear()\n\n    def configure_optimizers(self):\n        optimizer = torch.optim.AdamW(make_parameter_groups(self.model), lr=self.lr, weight_decay=self.weight_decay)\n        # scheduler = torch.optim.lr_scheduler.ReduceLROnPlateau(optimizer, mode='max', factor=0.5, patience=5,\n        #                                                        verbose=True)\n        return {\n            'optimizer': optimizer,\n            # 'lr_scheduler': {\n            #     'scheduler': scheduler,\n            #     'monitor': 'val_r_square',\n            # }\n        }\n\n    def on_train_epoch_end(self):\n        if self.trainer.sanity_checking:\n            return\n\n        y = torch.cat([x[1] for x in self.training_step_outputs]).cpu().numpy()\n        prob = torch.cat([x[0] for x in self.training_step_outputs]).detach().cpu().numpy()\n        weights = torch.cat([x[2] for x in self.training_step_outputs]).cpu().numpy()\n        # r2_training\n        train_r_square = r2_val(y, prob, weights)\n        self.log(\"train_r_square\", train_r_square, prog_bar=True, on_step=False, on_epoch=True)\n        self.training_step_outputs.clear()\n\n        epoch = self.trainer.current_epoch\n        metrics = {k: v.item() if isinstance(v, torch.Tensor) else v for k, v in self.trainer.logged_metrics.items()}\n        formatted_metrics = {k: f\"{v:.5f}\" for k, v in metrics.items()}\n        print(f\"Epoch {epoch}: {formatted_metrics}\")\n        \nclass custom_args():\n    def __init__(self):\n        self.usegpu = True\n        self.gpuid = 0\n        self.seed = 42\n        self.model = 'nn'\n        self.use_wandb = False\n        self.project = 'js-tabm-with-lags'\n        self.dname = \"./input_df/\"\n        self.loader_workers = 10   \n        self.bs = 8192\n        self.lr = 1e-3\n        self.weight_decay = 8e-4\n        self.n_cont_features = 84\n        self.n_cat_features = 5\n        self.n_classes = None\n        self.cat_cardinalities = [23, 10, 32, 40, 969]\n        self.patience = 7\n        self.max_epochs = 10\n        self.N_fold = 5\n\n\nmy_args = custom_args()\n\ndevice = torch.device('cuda:0' if torch.cuda.is_available() else 'cpu')\n\nmodel = NN.load_from_checkpoint('/kaggle/input/my-own-js/tabm_epochepoch03.ckpt').to(device)","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2025-01-13T20:05:44.589885Z","iopub.execute_input":"2025-01-13T20:05:44.590263Z","iopub.status.idle":"2025-01-13T20:05:44.723103Z","shell.execute_reply.started":"2025-01-13T20:05:44.590226Z","shell.execute_reply":"2025-01-13T20:05:44.722107Z"},"papermill":{"duration":0.253511,"end_time":"2025-01-13T14:46:09.893468","exception":false,"start_time":"2025-01-13T14:46:09.639957","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"id":"c5558b6e","cell_type":"code","source":"lags_ : pl.DataFrame | None = None\n\nlags_history = None\n\ndef predict_tabm(test: pl.DataFrame, lags: pl.DataFrame | None) -> pl.DataFrame | pd.DataFrame:\n    global lags_, lags_history\n    if lags is not None:\n        lags_ = lags\n    \n    for col in feature_cat + ['symbol_id', 'time_id']:\n        test = encode_column(test, col, category_mappings[col])\n\n    predictions = test.select(\n        'row_id',\n        pl.lit(0.0).alias('responder_6'),\n    )\n    \n    symbol_ids = test.select('symbol_id').to_numpy()[:, 0]\n\n    time_id = test.select(\"time_id\").to_numpy()[0]\n    timie_id_array = test.select(\"time_id\").to_numpy()[:, 0]\n    \n    \n    if time_id == 0:\n        lags = lags.with_columns(pl.col('time_id').cast(pl.Int64))\n        lags = lags.with_columns(pl.col('symbol_id').cast(pl.Int64))\n    \n        lags_history = lags\n        lags = lags.filter(pl.col(\"time_id\") == 0)\n        \n        \n        test = test.join(lags, on=[\"time_id\", \"symbol_id\"],  how=\"left\")\n    else:\n        lags = lags_history.filter(pl.col(\"time_id\") == time_id)\n        test = test.join(lags, on=[\"time_id\", \"symbol_id\"],  how=\"left\")\n\n    \n    test = test.with_columns([\n        pl.col(col).fill_null(0) for col in feature_list + [f\"responder_{idx}_lag_1\" for idx in range(9)] \n    ])\n\n    test = standardize(test, std_feature, means, stds)\n\n\n    X_test = test[feature_test].to_numpy()\n    X_test_tensor = torch.tensor(X_test, dtype=torch.float32).to(device)\n\n    symbol_tensor = torch.tensor(symbol_ids, dtype=torch.float32).to(device)\n    time_tensor = torch.tensor(timie_id_array, dtype=torch.float32).to(device)\n    X_cat = X_test_tensor[:, [9, 10, 11]]\n    X_cont = X_test_tensor[:, [i for i in range(X_test_tensor.shape[1]) if i not in [9, 10, 11]]]\n    # X_cont = X_cont + torch.randn_like(X_cont) * 0.02\n\n    X_cat = (torch.concat([X_cat, symbol_tensor.unsqueeze(-1), time_tensor.unsqueeze(-1)], axis=1)).to(torch.int64)\n    \n    model.eval()\n    with torch.no_grad():\n        \n        outputs = model(X_cont, X_cat)\n        # Assuming the model outputs a tensor of shape (batch_size, 1)\n        preds = outputs.squeeze(-1).cpu().numpy()\n        preds = preds.mean(1)\n    \n    predictions = \\\n    test.select('row_id').\\\n    with_columns(\n        pl.Series(\n            name   = 'responder_6', \n            values = np.clip(preds, a_min = -5, a_max = 5),\n            dtype  = pl.Float64,\n        )\n    )\n\n    return predictions","metadata":{"execution":{"iopub.status.busy":"2025-01-13T20:05:44.724422Z","iopub.execute_input":"2025-01-13T20:05:44.724784Z","iopub.status.idle":"2025-01-13T20:05:44.739694Z","shell.execute_reply.started":"2025-01-13T20:05:44.724749Z","shell.execute_reply":"2025-01-13T20:05:44.738389Z"},"papermill":{"duration":0.020739,"end_time":"2025-01-13T14:46:09.921336","exception":false,"start_time":"2025-01-13T14:46:09.900597","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"id":"2258a254","cell_type":"markdown","source":"## JS Ridge baseline (yunsuxiaozi)\n\n`LB: 0.0042` https://www.kaggle.com/code/yunsuxiaozi/js-ridge-baseline?scriptVersionId=202938352","metadata":{"papermill":{"duration":0.005844,"end_time":"2025-01-13T14:46:09.933751","exception":false,"start_time":"2025-01-13T14:46:09.927907","status":"completed"},"tags":[]}},{"id":"fad987fd","cell_type":"code","source":"def load_from_dill(model_name, model_path=None, file_ext='.dill'):\n    \"\"\"\n    Load a model from a dill file\n    \n    Args:\n        model_name: Name of the model file (without extension)\n        model_path: Directory path containing the model file\n        file_ext: File extension (default: '.dill')\n        \n    Returns:\n        Loaded model object\n    \"\"\"\n    model_object = None\n    # Open and load the model file using dill\n    with open(f\"{model_path}/{model_name}{file_ext}\", \"rb\") as file_handle:\n        model_object = dill.load(file_handle)\n    return model_object\n\n# Load pre-trained Ridge Regression model\nrdg = load_from_dill(\n    model_name='Ridge', \n    model_path=\"/kaggle/input/jsridgev01011635\"\n)\n\ndef predict_ridge(test, lags):\n    \"\"\"\n    Make predictions using Ridge Regression model\n    \n    Args:\n        test: DataFrame containing test data\n        lags: DataFrame containing lagged features (unused in this function)\n        \n    Returns:\n        DataFrame with predictions\n    \"\"\"\n    # Select the 79 numerical features\n    cols = [f'feature_{i:02}' for i in range(79)]\n\n    # Initialize predictions DataFrame with row_id and placeholder predictions\n    predictions = test.select(\n        'row_id',\n        pl.lit(0.0).alias('responder_6'),\n    )\n\n    # Generate predictions:\n    # 1. Select required features\n    # 2. Convert to pandas\n    # 3. Fill missing values with 3\n    # 4. Make predictions using Ridge model\n    test_preds = rdg.predict(test[cols].to_pandas().fillna(3).values)\n\n    # Add predictions to result DataFrame\n    predictions = predictions.with_columns(pl.Series('responder_6', test_preds.ravel()))\n\n    return predictions","metadata":{"execution":{"iopub.status.busy":"2025-01-13T20:05:44.741286Z","iopub.execute_input":"2025-01-13T20:05:44.741761Z","iopub.status.idle":"2025-01-13T20:05:44.756585Z","shell.execute_reply.started":"2025-01-13T20:05:44.741699Z","shell.execute_reply":"2025-01-13T20:05:44.755624Z"},"papermill":{"duration":0.01851,"end_time":"2025-01-13T14:46:09.958277","exception":false,"start_time":"2025-01-13T14:46:09.939767","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"id":"069038d3","cell_type":"markdown","source":"## CNN_LSTM Baseline (kayeyeye)","metadata":{"papermill":{"duration":0.006255,"end_time":"2025-01-13T14:46:09.970856","exception":false,"start_time":"2025-01-13T14:46:09.964601","status":"completed"},"tags":[]}},{"id":"4e22d8bc","cell_type":"code","source":"CNN_LSTM_CONFIG = {\n    \"seq_length\": 20,  # Length of input sequences for the model\n    \"batch_size\": 512,  # Number of samples per gradient update\n    \"learning_rate\": 0.0001,  # Learning rate for the optimizer\n    \"num_epochs\": 12,  # Number of training epochs\n    \"num_workers\": 4,  # Number of subprocesses to use for data loading\n    \"num_heads\": 8, \n    \"num_hash_buckets\": 64,\n    \"pin_memory\": True,  # Whether to pin memory for faster data transfer to GPU\n    \"prefetch_factor\": 16,  # Number of batches to prefetch\n    \"dropout_rate\": 0.2,  # Dropout rate for regularization\n    \"input_channels\": 79,  # Number of input features\n    \"output_features\": 1,  # Number of output features after CNN\n    \"lstm_hidden_size\": 128,  # Hidden size for LSTM\n    \"embedding_dim\": 16,  # Dimension of the embedding for symbol_id\n    \"weight_decay\": 0.0001, \n    # \"num_symbols\": 39,  # Number of unique symbol_ids (0 to 38)\n    \"feature_columns\": [f'feature_{i:02d}' for i in range(79)],  # List of feature column names\n    \"target_column\": \"responder_6\",  # Name of the target variable\n    # \"data_path\": \"/kaggle/input/jane-street-real-time-market-data-forecasting/train.parquet/\",  # Path to the dataset\n    # \"dataset_save_path\": \"/kaggle/working/processed_dataset.parquet\",  # Path to save the processed dataset\n    # \"checkpoint_dir\": \"/kaggle/working/models\",  # Path to save the trained model\n}\nfeature_columns = CNN_LSTM_CONFIG['feature_columns']\ntarget_column = CNN_LSTM_CONFIG['target_column']\nseq_length = CNN_LSTM_CONFIG['seq_length']\n\nclass TimeSeriesDataset(Dataset):\n    def __init__(self, data, seq_length, feature_columns, target_column):\n        # print('hello')\n        #unique_symbol_ids = data['symbol_id'].unique()  # Get unique symbol_ids\n        #sorted_symbol_ids = np.sort(unique_symbol_ids)\n        self.data = torch.tensor(data.select(feature_columns).to_numpy(), dtype=torch.float32)\n        if target_column is not None and target_column in data.columns:\n            self.targets = torch.tensor(data.select(target_column).to_numpy().flatten(), dtype=torch.float32)\n        else:\n            self.targets = None\n        self.symbol_ids = torch.tensor(data.select('symbol_id').to_numpy().flatten(), dtype=torch.long)\n        # print(\"hi\")\n        # # Ensure that the sequence length does not exceed the available data length\n        # if seq_length > len(self.data):\n        #     raise ValueError(\"Sequence length must be less than or equal to the length of the data.\")\n\n        self.seq_length = seq_length\n\n    def __len__(self):\n        return len(self.data) - self.seq_length + 1\n\n    def __getitem__(self, idx):\n        seq = self.data[idx:idx + self.seq_length]  # Shape: (seq_length, num_features)\n        if self.targets is not None:\n            target = self.targets[idx + self.seq_length - 1]  # Shape: (1,)\n        else:\n            target = None\n        symbol_ids = self.symbol_ids[idx:idx + self.seq_length]  # Get corresponding symbol_ids\n        # Ensure that seq and symbol_ids are of the same length\n        # if len(seq) != self.seq_length or len(symbol_ids) != self.seq_length:\n        #     raise ValueError(f\"Sequence length mismatch: seq length {len(seq)}, symbol_ids length {len(symbol_ids)}\")\n        if self.targets is not None:\n            return seq.permute(1, 0), target.unsqueeze(0), symbol_ids  # Return (num_features, seq_length), (1,), (seq_length,)\n        else:\n            return seq.permute(1, 0), target, symbol_ids  # Return (num_features, seq_length), (1,), (seq_length,)\n        ","metadata":{"execution":{"iopub.status.busy":"2025-01-13T20:05:44.758352Z","iopub.execute_input":"2025-01-13T20:05:44.758889Z","iopub.status.idle":"2025-01-13T20:05:44.771625Z","shell.execute_reply.started":"2025-01-13T20:05:44.758843Z","shell.execute_reply":"2025-01-13T20:05:44.770603Z"},"papermill":{"duration":0.018114,"end_time":"2025-01-13T14:46:09.995163","exception":false,"start_time":"2025-01-13T14:46:09.977049","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"id":"6fdfd189","cell_type":"code","source":"class AddNorm(nn.Module):\n    def __init__(self, size, dropout=0.1):\n        super(AddNorm, self).__init__()\n        self.norm = nn.LayerNorm(size)\n        self.dropout = nn.Dropout(dropout)\n\n    def forward(self, x, residual):\n        return self.norm(x + self.dropout(residual))\n\nclass GatedResidualNetwork(nn.Module):\n    def __init__(self, input_size):\n        super(GatedResidualNetwork, self).__init__()\n        self.linear1 = nn.Linear(input_size, input_size)\n        self.linear2 = nn.Linear(input_size, input_size)\n        self.activation = nn.ReLU()\n\n    def forward(self, x):\n        gate = torch.sigmoid(self.linear2(x))  # Gating mechanism\n        residual = self.linear1(x)  # Linear transformation\n        return gate * residual + x  # Residual connection\n\nclass AttCNNBiLSTM(nn.Module):\n    def __init__(self, input_channels, lstm_hidden_size, output_features, num_hash_buckets, embedding_dim, num_heads, dropout_rate=0.2):\n        super(AttCNNBiLSTM, self).__init__()\n        \n        # Embedding layer for symbol_id using hashing trick\n        self.num_hash_buckets = num_hash_buckets\n        self.embedding = nn.Embedding(num_hash_buckets, embedding_dim)  \n        \n        # Convolutional layers\n        self.conv1 = nn.Conv1d(in_channels=input_channels + embedding_dim, out_channels=128, kernel_size=3, padding=1)\n        self.bn1 = nn.BatchNorm1d(128)\n        #self.pool1 = nn.MaxPool1d(kernel_size=2)\n        self.dropout1 = nn.Dropout(dropout_rate)\n\n        self.conv2 = nn.Conv1d(in_channels=128, out_channels=64, kernel_size=3, padding=1)\n        self.bn2 = nn.BatchNorm1d(64)\n        #self.pool2 = nn.MaxPool1d(kernel_size=2)\n        self.dropout2 = nn.Dropout(dropout_rate)\n        \n        self.conv3 = nn.Conv1d(in_channels=64, out_channels=32, kernel_size=3, padding=1)\n        self.bn3 = nn.BatchNorm1d(32)\n        #self.pool3 = nn.MaxPool1d(kernel_size=2)\n        self.dropout3 = nn.Dropout(dropout_rate)\n        \n        # Additional GRN after CNN layers\n        #self.grn_cnn_to_lstm = GatedResidualNetwork(32)  # Input size matches the output of the last CNN layer\n        #self.add_norm_cnn_to_lstm = AddNorm(32)  # Add & Norm after GRN        \n\n        # LSTM layers\n        self.lstm = nn.LSTM(input_size=32, hidden_size=lstm_hidden_size, num_layers=2, batch_first=True, bidirectional=True)\n        self.dropout_lstm = nn.Dropout(dropout_rate)\n\n        # Multi-Head Attention\n        self.multihead_attn = nn.MultiheadAttention(embed_dim=lstm_hidden_size * 2, num_heads=num_heads, dropout=0.2, add_zero_attn=False)\n\n        # Gated Residual Network after Attention\n        self.grn1 = GatedResidualNetwork(lstm_hidden_size * 2)  # Input size matches the output of the attention layer\n        self.add_norm_attn = AddNorm(lstm_hidden_size * 2)  # Add & Norm after GRN\n\n        # Fully connected layer for prediction\n        #self.fc1 = nn.Linear(lstm_hidden_size * 2, 64)  # First fully connected layer\n        #self.fc2 = nn.Linear(64, 16)  # Second fully connected layer\n        self.fc_output = nn.Linear(lstm_hidden_size * 2, output_features)  # Final output layer\n\n    def forward(self, x, symbol_ids, return_features=False):\n        # Hashing trick to map symbol_ids to indices\n        hashed_indices = torch.fmod(symbol_ids, self.num_hash_buckets)  # Hashing trick\n        \n        # Embedding for hashed symbol_ids\n        embedded_symbols = self.embedding(hashed_indices)  # Shape: (batch_size, seq_length, embedding_dim)\n        embedded_symbols = embedded_symbols.permute(0, 2, 1)  # Shape: (batch_size, embedding_dim, seq_length)\n\n        # Concatenate embeddings with features\n        x = torch.cat((x, embedded_symbols), dim=1)  # Shape: (batch_size, input_channels + embedding_dim, seq_length)\n\n        # Apply Convolutional Layers\n        x = F.relu(self.bn1(self.conv1(x)))\n        #x = self.pool1(x)\n        x = self.dropout1(x)\n\n        x = F.relu(self.bn2(self.conv2(x)))\n        #x = self.pool2(x)\n        x = self.dropout2(x)\n        \n        x = F.relu(self.bn3(self.conv3(x)))\n        #x = self.pool3(x)\n        x = self.dropout3(x)\n        \n        # Apply Gated Residual Network\n        #x = self.grn_cnn_to_lstm(x.permute(0, 2, 1))  # Shape: (batch_size, seq_length, num_channels)\n        #x = x.permute(0, 2, 1)  # Shape: (batch_size, num_channels, seq_length)\n        \n        # Prepare for LSTM\n        x = x.permute(0, 2, 1)  # Shape: (batch_size, seq_length, num_channels)\n        \n        # Apply Add & Norm\n        #x = self.add_norm_cnn_to_lstm(x, x)  # Use the same x for residual connection\n\n        # Apply LSTM\n        lstm_out, _ = self.lstm(x)\n        lstm_out = self.dropout_lstm(lstm_out)\n\n        # Multi-Head Attention\n        attn_out, _ = self.multihead_attn(lstm_out, lstm_out, lstm_out)\n\n        # Apply Gated Residual Network and Add & Norm\n        grn_out = self.grn1(attn_out.mean(dim=1))  # Shape: (batch_size, 128)\n        grn_out = self.add_norm_attn(grn_out, attn_out.mean(dim=1))  # Apply Add & Norm\n\n        if return_features:\n            return grn_out\n\n        # Fully connected layers\n        #fc_out = F.relu(self.fc1(grn_out))  # First fully connected layer\n        #fc_out = F.relu(self.fc2(fc_out))  # Second fully connected layer\n\n        # Fully connected layer for prediction\n        output = self.fc_output(grn_out)  # Final output layer\n        return output\n","metadata":{"execution":{"iopub.status.busy":"2025-01-13T20:05:44.774825Z","iopub.execute_input":"2025-01-13T20:05:44.775231Z","iopub.status.idle":"2025-01-13T20:05:44.796126Z","shell.execute_reply.started":"2025-01-13T20:05:44.775196Z","shell.execute_reply":"2025-01-13T20:05:44.795133Z"},"papermill":{"duration":0.022317,"end_time":"2025-01-13T14:46:10.023544","exception":false,"start_time":"2025-01-13T14:46:10.001227","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"id":"c9ec101e","cell_type":"code","source":"def evaluate_CNN_LSTM_model(model, data_loader, device):\n    model.eval()\n    model.to(device)\n    all_predictions = []\n    # all_targets = []\n    \n    with torch.no_grad():\n        for batch_X, batch_y, symbol_ids in data_loader:\n            batch_X = batch_X.to(device)\n            # batch_y = batch_y.to(device)\n            symbol_ids = symbol_ids.to(device)\n            outputs = model(batch_X, symbol_ids)  # Get model predictions\n            all_predictions.append(outputs.cpu().numpy())\n            # all_targets.append(batch_y.cpu().numpy())\n    \n    # Concatenate all predictions and targets\n    all_predictions = np.concatenate(all_predictions)\n    # all_targets = np.concatenate(all_targets)\n\n    return all_predictions","metadata":{"execution":{"iopub.status.busy":"2025-01-13T20:05:44.797642Z","iopub.execute_input":"2025-01-13T20:05:44.798117Z","iopub.status.idle":"2025-01-13T20:05:44.810909Z","shell.execute_reply.started":"2025-01-13T20:05:44.798045Z","shell.execute_reply":"2025-01-13T20:05:44.809854Z"},"papermill":{"duration":0.014053,"end_time":"2025-01-13T14:46:10.043993","exception":false,"start_time":"2025-01-13T14:46:10.029940","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"id":"780cdba2","cell_type":"code","source":"import numpy as np\ndef predict_cnn_lstm(test, stocks, padding):\n    input_channels = CNN_LSTM_CONFIG['input_channels']\n    output_features = CNN_LSTM_CONFIG['output_features']\n    lstm_hidden_size = CNN_LSTM_CONFIG['lstm_hidden_size']\n    num_hash_buckets = CNN_LSTM_CONFIG['num_hash_buckets']\n    embedding_dim = CNN_LSTM_CONFIG['embedding_dim']\n    # num_symbols = CONFIG['num_symbols']\n    num_heads = CNN_LSTM_CONFIG['num_heads']\n    \n    device = torch.device(\"cuda:0\" if torch.cuda.is_available() else \"cpu\")  # Check if multiple GPUs are available\n    \n    model = AttCNNBiLSTM(input_channels, lstm_hidden_size, output_features, num_hash_buckets, embedding_dim, num_heads).to(device)\n    # Load the model state dictionary\n    model_path = '/kaggle/input/cnn-lstm-janestreet/model_epoch_12.pth'\n    model.load_state_dict(torch.load(model_path, map_location=device))\n    \n    # Move the model to the specified device\n    model.to(device)\n    train_loader = DataLoader(test, batch_size=CNN_LSTM_CONFIG['batch_size'], pin_memory=True, num_workers=4, shuffle=False)\n    # Check for GPU availability\n    device = torch.device('cuda:0' if torch.cuda.is_available() else 'cpu')\n\n    # Evaluate the model \n    predictions = evaluate_model(model, train_loader, device)\n\n    # Arrange the predictions as per the modified test set\n    final_predictions = []\n    extra_padding = [np.nan]*19\n    last = 0\n    for i in range(len(stocks)):\n        final_prediction= final_prediction+extra_padding\n        final_predictions = final_predictions +  predictions[last:last+padding[i]]\n        last = last+padding[i]\n        \n    predictions = predictions.with_columns(pl.Series('responder_6', final_predictions.ravel()))\n    predictions = preditions.select(\n        [\n            pl.col(col).fill_null(strategy=\"forward\").fill_null(strategy=\"backward\").alias(col)\n            for col in predictions.columns\n        ]\n    )\n    \n    return predictions","metadata":{"execution":{"iopub.status.busy":"2025-01-13T20:05:44.812469Z","iopub.execute_input":"2025-01-13T20:05:44.812815Z","iopub.status.idle":"2025-01-13T20:05:44.823838Z","shell.execute_reply.started":"2025-01-13T20:05:44.812782Z","shell.execute_reply":"2025-01-13T20:05:44.822791Z"},"papermill":{"duration":0.015986,"end_time":"2025-01-13T14:46:10.066444","exception":false,"start_time":"2025-01-13T14:46:10.050458","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"id":"31e6a676","cell_type":"code","source":"schema = {\n    \"row_id\": pl.Int64,\n    \"date_id\": pl.Int16,\n    \"time_id\": pl.Int16,\n    \"symbol_id\": pl.Int8,\n    \"weight\": pl.Float32,\n    \"is_scored\": pl.Boolean,\n    \"feature_00\": pl.Float32,\n    \"feature_01\": pl.Float32,\n    \"feature_02\": pl.Float32,\n    \"feature_03\": pl.Float32,\n    \"feature_04\": pl.Float32,\n    \"feature_05\": pl.Float32,\n    \"feature_06\": pl.Float32,\n    \"feature_07\": pl.Float32,\n    \"feature_08\": pl.Float32,\n    \"feature_09\": pl.Float64,\n    \"feature_10\": pl.Float64,\n    \"feature_11\": pl.Float64,\n    \"feature_12\": pl.Float32,\n    \"feature_13\": pl.Float32,\n    \"feature_14\": pl.Float32,\n    \"feature_15\": pl.Float32,\n    \"feature_16\": pl.Float32,\n    \"feature_17\": pl.Float32,\n    \"feature_18\": pl.Float32,\n    \"feature_19\": pl.Float32,\n    \"feature_20\": pl.Float32,\n    \"feature_21\": pl.Float32,\n    \"feature_22\": pl.Float32,\n    \"feature_23\": pl.Float32,\n    \"feature_24\": pl.Float32,\n    \"feature_25\": pl.Float32,\n    \"feature_26\": pl.Float32,\n    \"feature_27\": pl.Float32,\n    \"feature_28\": pl.Float32,\n    \"feature_29\": pl.Float32,\n    \"feature_30\": pl.Float32,\n    \"feature_31\": pl.Float32,\n    \"feature_32\": pl.Float32,\n    \"feature_33\": pl.Float32,\n    \"feature_34\": pl.Float32,\n    \"feature_35\": pl.Float32,\n    \"feature_36\": pl.Float32,\n    \"feature_37\": pl.Float32,\n    \"feature_38\": pl.Float32,\n    \"feature_39\": pl.Float32,\n    \"feature_40\": pl.Float32,\n    \"feature_41\": pl.Float32,\n    \"feature_42\": pl.Float32,\n    \"feature_43\": pl.Float32,\n    \"feature_44\": pl.Float32,\n    \"feature_45\": pl.Float32,\n    \"feature_46\": pl.Float32,\n    \"feature_47\": pl.Float32,\n    \"feature_48\": pl.Float32,\n    \"feature_49\": pl.Float32,\n    \"feature_50\": pl.Float32,\n    \"feature_51\": pl.Float32,\n    \"feature_52\": pl.Float32,\n    \"feature_53\": pl.Float32,\n    \"feature_54\": pl.Float32,\n    \"feature_55\": pl.Float32,\n    \"feature_56\": pl.Float32,\n    \"feature_57\": pl.Float32,\n    \"feature_58\": pl.Float32,\n    \"feature_59\": pl.Float32,\n    \"feature_60\": pl.Float32,\n    \"feature_61\": pl.Float32,\n    \"feature_62\": pl.Float32,\n    \"feature_63\": pl.Float32,\n    \"feature_64\": pl.Float32,\n    \"feature_65\": pl.Float32,\n    \"feature_66\": pl.Float32,\n    \"feature_67\": pl.Float32,\n    \"feature_68\": pl.Float32,\n    \"feature_69\": pl.Float32,\n    \"feature_70\": pl.Float32,\n    \"feature_71\": pl.Float32,\n    \"feature_72\": pl.Float32,\n    \"feature_73\": pl.Float32,\n    \"feature_74\": pl.Float32,\n    \"feature_75\": pl.Float32,\n    \"feature_76\": pl.Float32,\n    \"feature_77\": pl.Float32,\n    \"feature_78\": pl.Float32\n}\n\n# Function to generate random/default values based on the data type\ndef generate_random_value(dtype):\n    if dtype == pl.Int64:\n        return np.random.randint(1, 1000)\n    elif dtype == pl.Int16:\n        return np.random.randint(1, 100)\n    elif dtype == pl.Int8:\n        return np.random.randint(1, 50)\n    elif dtype == pl.Float32:\n        return np.float32(np.random.uniform(0.0, 100.0))\n    elif dtype == pl.Float64:\n        return np.random.uniform(0.0, 100.0)\n    elif dtype == pl.Boolean:\n        return np.random.choice([True, False])\n    else:\n        return None\n\n# Create a dictionary to hold the new data\nnew_data = {}\n\n# Loop through each column in the schema\nfor col, dtype in schema.items():\n    new_data[col] = [generate_random_value(dtype) for _ in range(25)]\n\n# Create the new DataFrame\nnew_df = pl.DataFrame(new_data)\n","metadata":{"execution":{"iopub.status.busy":"2025-01-13T20:05:44.825190Z","iopub.execute_input":"2025-01-13T20:05:44.825580Z","iopub.status.idle":"2025-01-13T20:05:44.855917Z","shell.execute_reply.started":"2025-01-13T20:05:44.825546Z","shell.execute_reply":"2025-01-13T20:05:44.854980Z"},"papermill":{"duration":0.042869,"end_time":"2025-01-13T14:46:10.115367","exception":false,"start_time":"2025-01-13T14:46:10.072498","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"id":"640d0ebf","cell_type":"markdown","source":"# Ensemble notebook","metadata":{"papermill":{"duration":0.006023,"end_time":"2025-01-13T14:46:10.128050","exception":false,"start_time":"2025-01-13T14:46:10.122027","status":"completed"},"tags":[]}},{"id":"152a7642","cell_type":"code","source":"def predict(test: pl.DataFrame, lags: pl.DataFrame | None) -> pl.DataFrame | pd.DataFrame:\n    \"\"\"\n    Make ensemble predictions combining all three models:\n    - Neural Network + XGBoost ensemble\n    - Starter ensemble (LightGBM + CatBoost + XGBoost)\n    - Ridge Regression\n    \n    Args:\n        test: DataFrame containing test data\n        lags: DataFrame containing lagged features\n        \n    Returns:\n        DataFrame with final ensemble predictions\n    \"\"\"\n    # Check to see if we are going to be using the CNN LSTM model or not\n    # print(\"1\")\n    test2 = test.sort(by=[\"symbol_id\",\"date_id\",\"time_id\"])\n    test2 = test2.select(\n        [\n            pl.col(col).fill_null(strategy=\"forward\").fill_null(strategy=\"backward\").alias(col)\n            for col in test2.columns\n        ]\n    )\n    # print(\"1\")\n    stocks = test2[\"symbol_id\"].unique()\n    stocks_len = []\n    # grouped = test2.group_by(\"symbol_id\")\n    abc=0\n    # print(\"1\")\n    CLtest= TimeSeriesDataset(new_df, seq_length, feature_columns, target_column)\n    # print(\"1\")\n    for stock in stocks:\n        \n        # print(stock)\n        filtered_df = test2.filter(pl.col(\"symbol_id\") == stock)\n        filtered_df = filtered_df.sort([\"date_id\",\"time_id\"])\n        # Count the number of rows in the filtered DataFrame\n        count = filtered_df.height\n        if count<20:\n            # Get predictions from each model/ensemble\n            # print(\"xgb\")\n            pd_nn_xgb = predict_nn_xgb(test, lags).to_pandas() # Neural Network + XGBoost ensemble\n            # print(\"r\")\n            pd_ridge = predict_ridge(test, lags).to_pandas()        # Ridge Regression\n            # print(\"nn\")\n            pd_tabm = predict_tabm(test, lags).to_pandas()  \n            \n            # Rename prediction columns to avoid conflicts\n            pd_nn_xgb = pd_nn_xgb.rename(columns={'responder_6': 'col_nn_xgb'})\n            pd_ridge = pd_ridge.rename(columns={'responder_6': 'col_ridge'})\n            pd_tabm  = pd_tabm.rename(columns={'responder_6': 'col_tabm'})\n        \n            # Merge all predictions based on row_id\n            pds = pd.merge(pd_nn_xgb, pd_ridge, on=['row_id'])\n            pds = pd.merge(pds, pd_tabm, on=['row_id'])\n        \n            e_weights = [0.55, 0.15, 0.3012]\n            # Create final weighted ensemble prediction:\n            pds['responder_6'] = (\n                pds['col_nn_xgb'] * e_weights[0] +\n                pds['col_ridge'] * e_weights[1] +\n                pds['col_tabm'] * e_weights[2]\n            )\n        \n            # Create final predictions DataFrame in required format\n            predictions = test.select('row_id', pl.lit(0.0).alias('responder_6'))\n            pred = pds['responder_6'].to_numpy()\n            predictions = predictions.with_columns(pl.Series('responder_6', pred.ravel()))\n            # print(\"done\")\n            if predictions.is_empty():\n                raise \"The predictions is empty\"\n            # print(\"hh\")\n            if isinstance(predictions, pl.DataFrame):\n                assert predictions.columns == ['row_id', 'responder_6']\n            elif isinstance(predictions, pd.DataFrame):\n                assert (predictions.columns == ['row_id', 'responder_6']).all()\n            else:\n                raise TypeError('The predict function must return a DataFrame')\n            # Confirm has as many rows as the test data.\n            assert len(predictions) == len(test)\n            # print(predictions)\n            return predictions\n        \n            \n        stock_len.append(count)\n        temp_dataset = TimeSeriesDataset(filtered_df, seq_length, feature_columns, target_column)\n        if abc==0:\n            CLtest.data = temp_datset.data\n            CLtest.targets = temp_datset.targets\n            CLtest.symbol_ids = temp_dataset.symbol_ids\n            abc =1\n            continue\n        CLtest.data = torch.cat((CLtest.data, temp_dataset.data),dim=0)\n        CLtest.targets = torch.cat((CLtest.targets, temp_dataset.targets),dim=0)\n        CLtest.symbol_ids = torch.cat((CLtest.symbol_ids, temp_dataset.symbol_ids),dim=0)\n\n    if min(stocks_len)>=20:\n        \n        paddings = [x-19 for x in stocks_len]\n        predictions_cnnlstm = test2.select('row_id',pl.lit(0.0).alias('responder_6'),)\n         # Get predictions from each model/ensemble\n        pd_nn_xgb = predict_nn_xgb(test, lags).to_pandas()     # Neural Network + XGBoost ensemble\n        pd_ridge = predict_ridge(test, lags).to_pandas()        # Ridge Regression\n        pd_tabm = predict_tabm(test, lags).to_pandas()  \n        pd_cnn_lstm = predict_cnn_lstm(CLtest, stocks, padding, predictions_cnnlstm).to_pandas()\n        \n        # Rename prediction columns to avoid conflicts\n        pd_nn_xgb = pd_nn_xgb.rename(columns={'responder_6': 'col_nn_xgb'})\n        pd_ridge = pd_ridge.rename(columns={'responder_6': 'col_ridge'})\n        pd_tabm  = pd_tabm.rename(columns={'responder_6': 'col_tabm'})\n        pd_cnn_lstm = pd_cnn_lstm.rename(columns={'responder_6': 'col_cnn_lstm'})\n    \n        # Merge all predictions based on row_id\n        pds = pd.merge(pd_nn_xgb, pd_ridge, on=['row_id'])\n        pds = pd.merge(pds, pd_tabm, on=['row_id'])\n        pds = pd.merge(pds, pd_cnn_lstm, on=[\"row_id\"])\n    \n        e_weights = [0.2, 0.1, 0.35, 0.35]\n        # Create final weighted ensemble prediction:\n        pds['responder_6'] = (\n            pds['col_nn_xgb'] * e_weights[0] +\n            pds['col_ridge'] * e_weights[1] +\n            pds['col_tabm'] * e_weights[2] +\n            pds[\"col_cnn_lstm\"] * e_weights[3]\n        )\n    \n        # Create final predictions DataFrame in required format\n        predictions = test.select('row_id', pl.lit(0.0).alias('responder_6'))\n        pred = pds['responder_6'].to_numpy()\n        predictions = predictions.with_columns(pl.Series('responder_6', pred.ravel()))\n        if predictions.is_empty():\n            raise \"The predictions is empty\"\n        # print(\"hh\")\n        if isinstance(predictions, pl.DataFrame):\n            assert predictions.columns == ['row_id', 'responder_6']\n        elif isinstance(predictions, pd.DataFrame):\n            assert (predictions.columns == ['row_id', 'responder_6']).all()\n        else:\n            raise TypeError('The predict function must return a DataFrame')\n        # Confirm has as many rows as the test data.\n        assert len(predictions) == len(test)\n        # print(\"hh\")\n        return predictions\n    \n        ","metadata":{"_kg_hide-input":false,"_kg_hide-output":false,"execution":{"iopub.status.busy":"2025-01-13T20:05:44.857485Z","iopub.execute_input":"2025-01-13T20:05:44.857838Z","iopub.status.idle":"2025-01-13T20:05:44.880464Z","shell.execute_reply.started":"2025-01-13T20:05:44.857802Z","shell.execute_reply":"2025-01-13T20:05:44.879313Z"},"papermill":{"duration":0.025276,"end_time":"2025-01-13T14:46:10.159411","exception":false,"start_time":"2025-01-13T14:46:10.134135","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"id":"6d83b72b","cell_type":"code","source":"import kaggle_evaluation.jane_street_inference_server\ninference_server = kaggle_evaluation.jane_street_inference_server.JSInferenceServer(predict)\n\nif os.getenv('KAGGLE_IS_COMPETITION_RERUN'):\n    inference_server.serve()\nelse:\n    inference_server.run_local_gateway(\n        (\n            '/kaggle/input/jane-street-real-time-market-data-forecasting/test.parquet',\n            '/kaggle/input/jane-street-real-time-market-data-forecasting/lags.parquet',\n        )\n    )","metadata":{"execution":{"iopub.status.busy":"2025-01-13T20:08:23.189352Z","iopub.execute_input":"2025-01-13T20:08:23.189801Z","iopub.status.idle":"2025-01-13T20:08:23.436954Z","shell.execute_reply.started":"2025-01-13T20:08:23.189765Z","shell.execute_reply":"2025-01-13T20:08:23.435498Z"},"papermill":{"duration":1.108877,"end_time":"2025-01-13T14:46:11.274433","exception":false,"start_time":"2025-01-13T14:46:10.165556","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"id":"7b866b10","cell_type":"code","source":"","metadata":{"papermill":{"duration":0.006525,"end_time":"2025-01-13T14:46:11.287902","exception":false,"start_time":"2025-01-13T14:46:11.281377","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null}]}