{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":84493,"databundleVersionId":9871156,"sourceType":"competition"},{"sourceId":9801075,"sourceType":"datasetVersion","datasetId":6006872},{"sourceId":9806342,"sourceType":"datasetVersion","datasetId":6010899},{"sourceId":10319914,"sourceType":"datasetVersion","datasetId":6389338},{"sourceId":10320223,"sourceType":"datasetVersion","datasetId":6389553},{"sourceId":203900450,"sourceType":"kernelVersion"},{"sourceId":204479873,"sourceType":"kernelVersion"},{"sourceId":207787842,"sourceType":"kernelVersion"},{"sourceId":213144305,"sourceType":"kernelVersion"},{"sourceId":214892435,"sourceType":"kernelVersion"},{"sourceId":171905,"sourceType":"modelInstanceVersion","modelInstanceId":146319,"modelId":168862}],"dockerImageVersionId":30787,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv, pd.read_parquet )\nimport polars as pl\n\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom matplotlib import pyplot as plt\nfrom matplotlib.ticker import MaxNLocator, FormatStrFormatter, PercentFormatter\n\nimport os, gc\nfrom tqdm.auto import tqdm\nimport pickle # module to serialize and deserialize objects\nimport re # for Regular expression operations \n\nimport tensorflow as tf\nfrom tensorflow.keras import layers, models\nfrom tensorflow.keras.optimizers import Adam\n\nimport torch\nimport torch.nn as nn\nimport torch.nn.functional as F\nfrom torch.utils.data  import Dataset, DataLoader\nfrom pytorch_lightning import (LightningDataModule, LightningModule, Trainer)\nfrom pytorch_lightning.callbacks import EarlyStopping, ModelCheckpoint, Timer\n\nfrom sklearn.metrics import r2_score\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.ensemble import VotingRegressor\n\nimport lightgbm as lgb\nfrom lightgbm import LGBMRegressor\n\nfrom xgboost import XGBRegressor\nfrom catboost import CatBoostRegressor\n\nimport warnings\nwarnings.filterwarnings('ignore')\npd.options.display.max_columns = None\n\n\nimport torch.optim\nfrom torch.utils.data import Dataset, DataLoader, TensorDataset\n\nfrom sklearn.model_selection import train_test_split\n\nfrom sklearn.metrics import r2_score\nimport pandas as pd\nimport math\nimport numpy as np\nfrom tqdm import tqdm\nimport polars as pl\nfrom collections import OrderedDict\nimport sys\n\nimport warnings\nwarnings.filterwarnings(\"ignore\")\n\nimport kaggle_evaluation.jane_street_inference_server\n\nimport os\n\nimport joblib\n\nfrom pytorch_lightning import LightningModule","metadata":{"_uuid":"f573766f-0a4b-4a41-a873-d3e78e56afaf","_cell_guid":"8936e699-b090-4d2b-9bf2-b77f90fbefdb","trusted":true,"execution":{"iopub.status.busy":"2024-12-28T16:06:37.124702Z","iopub.execute_input":"2024-12-28T16:06:37.125030Z","iopub.status.idle":"2024-12-28T16:06:37.134335Z","shell.execute_reply.started":"2024-12-28T16:06:37.125005Z","shell.execute_reply":"2024-12-28T16:06:37.133339Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"ENSEMBLE_SOLUTIONS = ['SOLUTION_14','SOLUTION_5']\nOPTION,__WTS = 'option 91',[0.85, 0.15]\nmodel_5 = None\nxgb_model = None\nmodels = None\ninitialized = False ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T16:06:37.135794Z","iopub.execute_input":"2024-12-28T16:06:37.136077Z","iopub.status.idle":"2024-12-28T16:06:37.150589Z","shell.execute_reply.started":"2024-12-28T16:06:37.136050Z","shell.execute_reply":"2024-12-28T16:06:37.149660Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":" # 1. [JS Ridge baseline](https://www.kaggle.com/code/yunsuxiaozi/js-ridge-baseline) Lb=0.0026\n [yunsuxiaozi](https://www.kaggle.com/yunsuxiaozi)","metadata":{}},{"cell_type":"code","source":"if 'SOLUTION_5' in ENSEMBLE_SOLUTIONS:\n    \n    def predict_5(test, lags):\n        # Use global model_5\n        global model_5\n        cols=[f'feature_0{i}' if i<10 else f'feature_{i}' for i in range(79)]\n        predictions = test.select('row_id', pl.lit(0.0).alias('responder_6'))\n        test_preds = model_5.predict(test[cols].to_pandas().fillna(3).values)\n        predictions = predictions.with_columns(pl.Series('responder_6', test_preds.ravel()))\n        return predictions\n\nif 'SOLUTION_5' in ENSEMBLE_SOLUTIONS:\n    from sklearn.linear_model import BayesianRidge\n    import joblib\n    model_5 = joblib.load('/kaggle/input/jane-street-5-and-7_/other/default/1/ridge_model_5(1).pkl')\n    # model_5 = joblib.load('/kaggle/input/jane-street-5-and-7_/other/default/1/bayesian_ridge_model_7(1).pkl')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T16:06:37.163186Z","iopub.execute_input":"2024-12-28T16:06:37.163457Z","iopub.status.idle":"2024-12-28T16:06:37.178666Z","shell.execute_reply.started":"2024-12-28T16:06:37.163432Z","shell.execute_reply":"2024-12-28T16:06:37.177813Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 2.. [Jane Street RMF NN + XGB](https://www.kaggle.com/code/voix97/jane-street-rmf-nn-xgb), Lb=0.0076\n [Xiang Sheng](https://www.kaggle.com/voix97)","metadata":{}},{"cell_type":"code","source":"if 'SOLUTION_14' in ENSEMBLE_SOLUTIONS:    \n    \n    class CONFIG:\n        seed = 42\n        target_col = \"responder_6\"\n        # feature_cols = [\"symbol_id\", \"time_id\"] + [f\"feature_{idx:02d}\" for idx in range(79)]+ [f\"responder_{idx}_lag_1\" for idx in range(9)]\n        feature_cols = [f\"feature_{idx:02d}\" for idx in range(79)]+ [f\"responder_{idx}_lag_1\" for idx in range(9)]\n\n        model_paths = [\n            \"/kaggle/input/js-xs-nn-trained-model\",\n            # \"/kaggle/input/jx-xgb-1/XGB.pkl\",\n            '/kaggle/input/jx-xgb-forest-1/XGB_ensemble.pkl'\n        ]\n\nif 'SOLUTION_14' in ENSEMBLE_SOLUTIONS:\n    \n    valid = pl.scan_parquet(\n        f\"/kaggle/input/js24-preprocessing-create-lags/validation.parquet/\"\n    ).collect().to_pandas()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T16:06:37.179739Z","iopub.execute_input":"2024-12-28T16:06:37.180065Z","iopub.status.idle":"2024-12-28T16:06:38.023828Z","shell.execute_reply.started":"2024-12-28T16:06:37.180025Z","shell.execute_reply":"2024-12-28T16:06:38.023106Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"if 'SOLUTION_14' in ENSEMBLE_SOLUTIONS: \n    \n    xgb_model = None\n    model_path = CONFIG.model_paths[1]\n\n    # load xgb\n    # with open( model_path, \"rb\") as fp:\n    #     result = pickle.load(fp)\n    #     xgb_model = result[\"model\"]\n\n    \n    # load forest\n    with open(model_path, \"rb\") as fp:\n        saved_ensemble = pickle.load(fp)\n        forests_loaded = saved_ensemble[\"models\"]\n        features_forest_loaded = saved_ensemble[\"subsets_of_features\"]\n\n    xgb_feature_cols = [\"symbol_id\", \"time_id\"] + CONFIG.feature_cols\n\n    # Show model\n    #display(xgb_model)\n\nif 'SOLUTION_14' in ENSEMBLE_SOLUTIONS:\n    \n    # Custom R2 metric for validation\n    def r2_val(y_true, y_pred, sample_weight):\n        r2 = 1 - np.average((y_pred - y_true) ** 2, weights=sample_weight) / (np.average((y_true) ** 2, weights=sample_weight) + 1e-38)\n        return r2\n\n\n    class NN_14(LightningModule):\n        def __init__(self, input_dim, hidden_dims, dropouts, lr, weight_decay):\n            super().__init__()\n            self.save_hyperparameters()\n            layers = []\n            in_dim = input_dim\n            for i, hidden_dim in enumerate(hidden_dims):\n                layers.append(nn.BatchNorm1d(in_dim))\n                if i > 0:\n                    layers.append(nn.SiLU())\n                if i < len(dropouts):\n                    layers.append(nn.Dropout(dropouts[i]))\n                layers.append(nn.Linear(in_dim, hidden_dim))\n                # layers.append(nn.ReLU())\n                in_dim = hidden_dim\n            layers.append(nn.Linear(in_dim, 1))  \n            layers.append(nn.Tanh())\n            self.model = nn.Sequential(*layers)\n            self.lr = lr\n            self.weight_decay = weight_decay\n            self.validation_step_outputs = []\n\n        def forward(self, x):\n            return 5 * self.model(x).squeeze(-1)  \n\n        def training_step(self, batch):\n            x, y, w = batch\n            y_hat = self(x)\n            loss = F.mse_loss(y_hat, y, reduction='none') * w  \n            loss = loss.mean()\n            self.log('train_loss', loss, on_step=False, on_epoch=True, batch_size=x.size(0))\n            return loss\n\n        def validation_step(self, batch):\n            x, y, w = batch\n            y_hat = self(x)\n            loss = F.mse_loss(y_hat, y, reduction='none') * w\n            loss = loss.mean()\n            self.log('val_loss', loss, on_step=False, on_epoch=True, batch_size=x.size(0))\n            self.validation_step_outputs.append((y_hat, y, w))\n            return loss\n\n        def on_validation_epoch_end(self):\n            \"\"\"Calculate validation WRMSE at the end of the epoch.\"\"\"\n            y = torch.cat([x[1] for x in self.validation_step_outputs]).cpu().numpy()\n            if self.trainer.sanity_checking:\n                prob = torch.cat([x[0] for x in self.validation_step_outputs]).cpu().numpy()\n            else:\n                prob = torch.cat([x[0] for x in self.validation_step_outputs]).cpu().numpy()\n                weights = torch.cat([x[2] for x in self.validation_step_outputs]).cpu().numpy()\n                # r2_val\n                val_r_square = r2_val(y, prob, weights)\n                self.log(\"val_r_square\", val_r_square, prog_bar=True, on_step=False, on_epoch=True)\n            self.validation_step_outputs.clear()\n        def configure_optimizers(self):\n            optimizer = torch.optim.Adam(self.parameters(), lr=self.lr, weight_decay=self.weight_decay)\n            scheduler = torch.optim.lr_scheduler.ReduceLROnPlateau(optimizer, mode='min', factor=0.5, patience=5,\n                                                                   verbose=True)\n            return {\n                'optimizer': optimizer,\n                'lr_scheduler': {\n                    'scheduler': scheduler,\n                    'monitor': 'val_loss',\n                }\n            }\n\n        def on_train_epoch_end(self):\n            if self.trainer.sanity_checking:\n                return\n            epoch = self.trainer.current_epoch\n            metrics = {k: v.item() if isinstance(v, torch.Tensor) else v for k, v in self.trainer.logged_metrics.items()}\n            formatted_metrics = {k: f\"{v:.5f}\" for k, v in metrics.items()}\n            print(f\"Epoch {epoch}: {formatted_metrics}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T16:06:38.026251Z","iopub.execute_input":"2024-12-28T16:06:38.026792Z","iopub.status.idle":"2024-12-28T16:06:38.087114Z","shell.execute_reply.started":"2024-12-28T16:06:38.026750Z","shell.execute_reply":"2024-12-28T16:06:38.086102Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# load nn\n\nif 'SOLUTION_14' in ENSEMBLE_SOLUTIONS:\n    \n    N_folds = 5\n    models = []\n    for fold in range(N_folds):\n        checkpoint_path = f\"{CONFIG.model_paths[0]}/nn_{fold}.model\"\n        model_14 = NN_14.load_from_checkpoint(checkpoint_path)\n        models.append(model_14.to(\"cuda:0\"))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T16:06:38.088295Z","iopub.execute_input":"2024-12-28T16:06:38.088647Z","iopub.status.idle":"2024-12-28T16:06:38.381850Z","shell.execute_reply.started":"2024-12-28T16:06:38.088602Z","shell.execute_reply":"2024-12-28T16:06:38.380865Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"if 'SOLUTION_14' in ENSEMBLE_SOLUTIONS:    \n    \n    lags_ : pl.DataFrame | None = None\n\n    def predict_14(test: pl.DataFrame, lags: pl.DataFrame | None) -> pl.DataFrame | pd.DataFrame:\n        global lags_, xgb_model, models\n        if lags is not None:\n            lags_ = lags\n\n        predictions_14 = test.select(\n            'row_id',\n            pl.lit(0.0).alias('responder_6'),\n        )\n        symbol_ids = test.select('symbol_id').to_numpy()[:, 0]\n\n        if not lags is None:\n            lags = lags.group_by([\"date_id\", \"symbol_id\"], maintain_order=True).last() # pick up last record of previous date\n            test = test.join(lags, on=[\"date_id\", \"symbol_id\"],  how=\"left\")\n        else:\n            test = test.with_columns(\n                ( pl.lit(0.0).alias(f'responder_{idx}_lag_1') for idx in range(9) )\n            )\n\n        preds = np.zeros((test.shape[0],))\n        \n        # weights for xgb and nn\n        w1, w2 = 0.4, 0.6\n\n        # load X_test for XGB and convert to categorical\n        test_input = test[xgb_feature_cols].to_pandas()\n        test_input = test_input.fillna(method = 'ffill').fillna(0)\n        test_input[\"symbol_id\"] = pd.Categorical(test_input[\"symbol_id\"], categories=list(range(39)))\n\n        # load xgb forest\n        forest_preds_list = []\n        for m, feats in zip(forests_loaded, features_forest_loaded):\n            forest_preds_list.append(m.predict(test_input[feats]))\n\n        # Average predictions for xgb forest\n        ensemble_preds = np.mean(forest_preds_list, axis=0)\n\n        # add xgb forest preds\n        preds += ensemble_preds * w1\n\n        # load X_test for NN\n        test_input = test_input[CONFIG.feature_cols]\n        test_input = torch.FloatTensor(test_input.values).to(\"cuda:0\")\n\n        # add nn\n        with torch.no_grad():\n            for i, nn_model in enumerate(tqdm(models)):\n                nn_model.eval()\n                preds += (nn_model(test_input).cpu().numpy() / len(models)) * w2\n        print(f\"predict> preds.shape =\", preds.shape)\n\n        predictions_14 = \\\n        test.select('row_id').\\\n        with_columns(\n            pl.Series(\n                name   = 'responder_6', \n                values = np.clip(preds, a_min = -5, a_max = 5),\n                dtype  = pl.Float64,\n            )\n        )\n\n        # The predict function must return a DataFrame\n        #assert isinstance(predictions, pl.DataFrame | pd.DataFrame)\n        # with columns 'row_id', 'responer_6'\n        #assert list(predictions.columns) == ['row_id', 'responder_6']\n        # and as many rows as the test data.\n        #assert len(predictions) == len(test)\n\n        return predictions_14","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T16:21:02.919536Z","iopub.execute_input":"2024-12-28T16:21:02.920507Z","iopub.status.idle":"2024-12-28T16:21:02.931591Z","shell.execute_reply.started":"2024-12-28T16:21:02.920463Z","shell.execute_reply":"2024-12-28T16:21:02.930619Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def predict(test: pl.DataFrame, lags: pl.DataFrame | None) -> pl.DataFrame | pd.DataFrame:\n    global initialized, model_5, xgb_model, models\n\n    # Load model_5 (ridge) and model_14 (xgb+NN)\n    if not initialized:\n        # Load heavy models here\n        import pickle\n        import joblib\n\n        # Load model_5\n        model_5 = joblib.load('/kaggle/input/jane-street-5-and-7_/other/default/1/ridge_model_5(1).pkl')\n        # model_5 = joblib.load('/kaggle/input/jane-street-5-and-7_/other/default/1/bayesian_ridge_model_7(1).pkl')\n\n        # Load NN models\n        N_folds = 5\n        models = []\n        for fold in range(N_folds):\n            checkpoint_path = f\"{CONFIG.model_paths[0]}/nn_{fold}.model\"\n            model_nn = NN_14.load_from_checkpoint(checkpoint_path)\n            models.append(model_nn.to(\"cuda:0\"))\n\n        initialized = True  # Mark as loaded\n\n    # Now use predict_14 and predict_5 with the loaded models\n    pdA = predict_14(test, lags).to_pandas()\n    # pdB = predict_transformer(test, lags).to_pandas()\n    pdC = predict_5(test, lags).to_pandas()\n    pdA = pdA.rename(columns={'responder_6':'responder_A'})\n    # pdB = pdB.rename(columns={'responder_6':'responder_B'})\n    pdC = pdC.rename(columns={'responder_6':'responder_C'})\n    # pds = pd.merge(pdA, pdB, on=['row_id'])\n    pds = pd.merge(pdA, pdC, on=['row_id'])\n    pds['responder_6'] = pds['responder_A'] * __WTS[0] + pds['responder_C'] * __WTS[1]\n    \n    # Plot all columns in the ensemble DataFrame in one plot\n    # plt.figure()\n    # for column in pds.columns:\n    #     if column == 'row_id':\n    #         continue\n    #     plt.plot(pds[column], label=column)\n    # plt.title('A: 14 (NN+XGB), C: Ridge')\n    # plt.xlabel('Index')\n    # plt.ylabel('Value')\n    # plt.legend()\n    # plt.grid(True)\n    # plt.show()\n    \n    \n    predictions = test.select('row_id', pl.lit(0.0).alias('responder_6'))\n    pred = pds['responder_6'].to_numpy()\n    predictions = predictions.with_columns(pl.Series('responder_6', pred.ravel()))\n    # display(predictions)\n    return predictions","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T16:21:06.327280Z","iopub.execute_input":"2024-12-28T16:21:06.327652Z","iopub.status.idle":"2024-12-28T16:21:06.335158Z","shell.execute_reply.started":"2024-12-28T16:21:06.327624Z","shell.execute_reply":"2024-12-28T16:21:06.334252Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"inference_server = kaggle_evaluation.jane_street_inference_server.JSInferenceServer(predict)\n\nif os.getenv('KAGGLE_IS_COMPETITION_RERUN'):\n    inference_server.serve()\nelse:\n    inference_server.run_local_gateway(\n        (\n            '/kaggle/input/jane-street-real-time-market-data-forecasting/test.parquet',\n            '/kaggle/input/jane-street-real-time-market-data-forecasting/lags.parquet',\n        )\n    )","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T16:21:08.779121Z","iopub.execute_input":"2024-12-28T16:21:08.779891Z","iopub.status.idle":"2024-12-28T16:21:08.923211Z","shell.execute_reply.started":"2024-12-28T16:21:08.779855Z","shell.execute_reply":"2024-12-28T16:21:08.921209Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}