{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"codemirror_mode":{"name":"ipython","version":3},"file_extension":".py","mimetype":"text/x-python","name":"python","nbconvert_exporter":"python","pygments_lexer":"ipython3","version":"3.10.14"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":84493,"databundleVersionId":9871156,"sourceType":"competition"},{"sourceId":9801075,"sourceType":"datasetVersion","datasetId":6006872},{"sourceId":9806342,"sourceType":"datasetVersion","datasetId":6010899},{"sourceId":10277864,"sourceType":"datasetVersion","datasetId":6359705},{"sourceId":10304887,"sourceType":"datasetVersion","datasetId":6378806},{"sourceId":203900450,"sourceType":"kernelVersion"}],"dockerImageVersionId":30787,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true},"papermill":{"default_parameters":{},"duration":7.594014,"end_time":"2024-10-10T11:58:36.355301","environment_variables":{},"exception":null,"input_path":"__notebook__.ipynb","output_path":"__notebook__.ipynb","parameters":{},"start_time":"2024-10-10T11:58:28.761287","version":"2.6.0"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"### ℹ️ Info\n* **forked original great work kernels**\n    * https://www.kaggle.com/code/voix97/jane-street-rmf-inference-nn-xgb\n* **2024/12/26 My Additional**\n    * XGB local train modelx5 add. dataset is here.\n    * https://www.kaggle.com/datasets/hideyukizushi/als-e-105-pp0-30-xgb-5fold\n    ```\n    # My add XGB CV\n    Fold:0\t0.02815069772843637\n    Fold:1\t0.03025051553457081\n    Fold:2\t0.030019236957170792\n    Fold:3\t0.01810551744095168\n    Fold:4\t0.029192405102951957\n    ```\n* **2024/12/30 My Additional**\n    * XGB local train modelx5 add. dataset is here.\n    * https://www.kaggle.com/datasets/hideyukizushi/als-e-106-pp0-40-xgb-5fold\n    ```\n    # My add XGB CV\n    Fold:0    0.03925861557112098\n    Fold:1    0.03926324675193538\n    Fold:2    0.036001183736670606\n    Fold:3    0.03101361931294766\n    Fold:4    0.039566354720759755\n    ```","metadata":{}},{"cell_type":"markdown","source":"---\n---","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport polars as pl\nimport numpy as np\nimport os, gc\nfrom tqdm.auto import tqdm\nimport pickle\n\nimport torch\nimport torch.nn as nn\nimport torch.nn.functional as F\nfrom pytorch_lightning import (LightningDataModule, LightningModule, Trainer)\nfrom pytorch_lightning.callbacks import EarlyStopping, ModelCheckpoint, Timer\n\nfrom sklearn.metrics import r2_score\nfrom sklearn.model_selection import train_test_split\nfrom torch.utils.data import Dataset, DataLoader\nfrom xgboost import XGBRegressor\n\nimport warnings\nwarnings.filterwarnings('ignore')\npd.options.display.max_columns = None\n\nimport kaggle_evaluation.jane_street_inference_server","metadata":{"execution":{"iopub.status.busy":"2024-12-31T08:33:45.241519Z","iopub.execute_input":"2024-12-31T08:33:45.242439Z","iopub.status.idle":"2024-12-31T08:33:52.305638Z","shell.execute_reply.started":"2024-12-31T08:33:45.242404Z","shell.execute_reply":"2024-12-31T08:33:52.304681Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Configurations","metadata":{}},{"cell_type":"code","source":"# 設定クラス\nclass CONFIG:\n    seed = 286 # 乱数生成のシード値を指定。実験の再現性を高める\n    target_col = \"responder_6\" # 目的変数\n    # feature_cols = [\"symbol_id\", \"time_id\"] + [f\"feature_{idx:02d}\" for idx in range(79)]+ [f\"responder_{idx}_lag_1\" for idx in range(9)]\n    feature_cols = [f\"feature_{idx:02d}\" for idx in range(79)]+ [f\"responder_{idx}_lag_1\" for idx in range(9)] # モデルの入力として使用する特徴量のリスト\n    \n    model_paths = [\n        #\"/kaggle/input/js24-train-gbdt-model-with-lags-singlemodel/result.pkl\",\n        #\"/kaggle/input/js24-trained-gbdt-model/result.pkl\",\n        \"/kaggle/input/js-xs-nn-trained-model\", # nnモデル\n    ]","metadata":{"execution":{"iopub.status.busy":"2024-12-31T08:33:57.287427Z","iopub.execute_input":"2024-12-31T08:33:57.288234Z","iopub.status.idle":"2024-12-31T08:33:57.293046Z","shell.execute_reply.started":"2024-12-31T08:33:57.288199Z","shell.execute_reply":"2024-12-31T08:33:57.292133Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Load preprocessed data (to calculate CV)","metadata":{}},{"cell_type":"code","source":"valid = pl.scan_parquet(\n    f\"/kaggle/input/js24-preprocessing-create-lags/validation.parquet/\"\n).collect().to_pandas() # scan_parquetの方が効率的に読み込める。→DFに変換","metadata":{"execution":{"iopub.status.busy":"2024-12-31T08:33:59.518949Z","iopub.execute_input":"2024-12-31T08:33:59.519300Z","iopub.status.idle":"2024-12-31T08:34:01.852949Z","shell.execute_reply.started":"2024-12-31T08:33:59.519254Z","shell.execute_reply":"2024-12-31T08:34:01.851979Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"---\n# **《《《XGB Models》》》**\n---","metadata":{}},{"cell_type":"code","source":"xgb_model = None\nmodel_path = \"/kaggle/input/als-e-106-pp0-40-xgb-5fold/result0.pkl\"\nwith open( model_path, \"rb\") as fp:\n    result = pickle.load(fp)\n    xgb_model = result[\"model\"] # モデルオブジェクトを指定\n    xgb_model.set_params(\n        early_stopping_rounds=50,\n        gamma=0.4,\n        tree_method=\"hist\",\n        max_depth=5,\n        eval_metric='rmse',\n        learning_rate=0.05\n    )\n\nxgb_feature_cols = [\"symbol_id\", \"time_id\"] + CONFIG.feature_cols # XGBの特徴量指定\n\n# Show model\ndisplay(xgb_model)\n\n# 以下同様にモデル構築","metadata":{"execution":{"iopub.status.busy":"2024-12-31T08:34:11.580320Z","iopub.execute_input":"2024-12-31T08:34:11.581151Z","iopub.status.idle":"2024-12-31T08:34:11.602515Z","shell.execute_reply.started":"2024-12-31T08:34:11.581117Z","shell.execute_reply":"2024-12-31T08:34:11.601405Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"xgb_model1 = None\nmodel_path = \"/kaggle/input/als-e-106-pp0-40-xgb-5fold/result1.pkl\"\nwith open( model_path, \"rb\") as fp:\n    result = pickle.load(fp)\n    xgb_model1 = result[\"model\"]\n    xgb_model1.set_params(\n        early_stopping_rounds=50,\n        gamma=0.4,\n        tree_method=\"hist\",\n        max_depth=5,\n        eval_metric='rmse',\n        learning_rate=0.05\n    )\n\nxgb_feature_cols = [\"symbol_id\", \"time_id\"] + CONFIG.feature_cols\n\n# Show model\ndisplay(xgb_model1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T08:34:20.197974Z","iopub.execute_input":"2024-12-31T08:34:20.198378Z","iopub.status.idle":"2024-12-31T08:34:20.216695Z","shell.execute_reply.started":"2024-12-31T08:34:20.198342Z","shell.execute_reply":"2024-12-31T08:34:20.215677Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"xgb_model2 = None\nmodel_path = \"/kaggle/input/als-e-106-pp0-40-xgb-5fold/result2.pkl\"\nwith open( model_path, \"rb\") as fp:\n    result = pickle.load(fp)\n    xgb_model2 = result[\"model\"]\n    xgb_model2.set_params(\n        early_stopping_rounds=50,\n        gamma=0.4,\n        tree_method=\"hist\",\n        max_depth=5,\n        eval_metric='rmse',\n        learning_rate=0.05\n    )\n\nxgb_feature_cols = [\"symbol_id\", \"time_id\"] + CONFIG.feature_cols\n\n# Show model\ndisplay(xgb_model2)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T08:34:26.791798Z","iopub.execute_input":"2024-12-31T08:34:26.792136Z","iopub.status.idle":"2024-12-31T08:34:26.822351Z","shell.execute_reply.started":"2024-12-31T08:34:26.792107Z","shell.execute_reply":"2024-12-31T08:34:26.821568Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"xgb_model3 = None\nmodel_path = \"/kaggle/input/als-e-106-pp0-40-xgb-5fold/result3.pkl\"\nwith open( model_path, \"rb\") as fp:\n    result = pickle.load(fp)\n    xgb_model3 = result[\"model\"]\n    xgb_model3.set_params(\n        early_stopping_rounds=50,\n        gamma=0.4,\n        tree_method=\"hist\",\n        max_depth=5,\n        eval_metric='rmse',\n        learning_rate=0.05\n    )\n\nxgb_feature_cols = [\"symbol_id\", \"time_id\"] + CONFIG.feature_cols\n\n# Show model\ndisplay(xgb_model3)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T08:34:37.806866Z","iopub.execute_input":"2024-12-31T08:34:37.807178Z","iopub.status.idle":"2024-12-31T08:34:37.823907Z","shell.execute_reply.started":"2024-12-31T08:34:37.807155Z","shell.execute_reply":"2024-12-31T08:34:37.823052Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"xgb_model4 = None\nmodel_path = \"/kaggle/input/als-e-106-pp0-40-xgb-5fold/result4.pkl\"\nwith open( model_path, \"rb\") as fp:\n    result = pickle.load(fp)\n    xgb_model4 = result[\"model\"]\n    xgb_model4.set_params(\n        early_stopping_rounds=50,\n        gamma=0.4,\n        tree_method=\"hist\",\n        max_depth=5,\n        eval_metric='rmse',\n        learning_rate=0.05\n    )\n\nxgbxgb_model3ols = [\"symbol_id\", \"time_id\"] + CONFIG.feature_cols\n\n# Show model\ndisplay(xgb_model4)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T08:34:45.379778Z","iopub.execute_input":"2024-12-31T08:34:45.380598Z","iopub.status.idle":"2024-12-31T08:34:45.410925Z","shell.execute_reply.started":"2024-12-31T08:34:45.380563Z","shell.execute_reply":"2024-12-31T08:34:45.410100Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ------------------------------------------- #\n# Public XGB Model\n# ------------------------------------------- #\nxgb_model5 = None\nmodel_path = \"/kaggle/input/js-with-lags-trained-xgb/result.pkl\"\nwith open( model_path, \"rb\") as fp:\n    result = pickle.load(fp)\n    xgb_model5 = result[\"model\"]\n    xgb_model5.set_params(\n        early_stopping_rounds=50,\n        gamma=0.4,\n        tree_method=\"hist\",\n        max_depth=5,\n        eval_metric='rmse',\n        learning_rate=0.05\n    )\n\nxgb_feature_cols = [\"symbol_id\", \"time_id\"] + CONFIG.feature_cols\n\n# Show model\ndisplay(xgb_model5)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T08:34:52.315881Z","iopub.execute_input":"2024-12-31T08:34:52.316260Z","iopub.status.idle":"2024-12-31T08:34:52.374288Z","shell.execute_reply.started":"2024-12-31T08:34:52.316231Z","shell.execute_reply":"2024-12-31T08:34:52.373417Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"---\n# **《《《NN Models》》》**\n---","metadata":{}},{"cell_type":"code","source":"# カスタムのR2スコア計算関数\ndef r2_val(y_true, y_pred, sample_weight):\n    r2 = 1 - np.average((y_pred - y_true) ** 2, weights=sample_weight) / (np.average((y_true) ** 2, weights=sample_weight) + 1e-38) # R2スコア計算\n    return r2\n\n\nclass NN(LightningModule):\n    # NNモデルの定義\n    def __init__(self, input_dim, hidden_dims, dropouts, lr, weight_decay):\n        # ~引数~\n        # input_dim: 入力層の次元数\n        # hidden_dims: 隠れ層の次元数をリスト形式で指定\n        # dropouts: 各隠れ層のドロップアウト率をリスト形式で指定する引数\n        # lr: 学習率\n        # weight_decay: 正則化のための重み減衰\n        \n        super().__init__() # LightningModuleのコンストラクタを呼び出すためのコード\n        self.save_hyperparameters() # コンストラクタに渡された引数をハイパーパラメータとして保存\n        layers = [] # NNの各層を格納するための空のリスト\n        in_dim = input_dim\n        \n        for i, hidden_dim in enumerate(hidden_dims):\n            layers.append(nn.BatchNorm1d(in_dim)) # バッチ正規化層をlayersリストに追加\n            if i > 0: # 最初の隠れ層以降でのみ実行\n                layers.append(nn.SiLU()) # 活性化関数SiLUをlayersリストに追加\n            if i < len(dropouts): # ドロップアウト層を適用する条件分岐\n                layers.append(nn.Dropout(dropouts[i])) # ドロップアウト層をlayersリストに追加\n            layers.append(nn.Linear(in_dim, hidden_dim)) # 線形層をlayersリストに追加\n            # layers.append(nn.ReLU())\n            in_dim = hidden_dim # 次の層の入力次元数を設定\n        layers.append(nn.Linear(in_dim, 1))  # 出力層として、入力次元数から出力次元数1への線形層をlayersリストに追加\n        layers.append(nn.Tanh()) # 出力層の活性化関数としてTanhをlayersリストに追加\n        self.model = nn.Sequential(*layers) # リストに格納された層を順番に実行するnn.Sequentialモデルを作成し、self.modelに格納\n        self.lr = lr # コンストラクタに渡された学習率をself.lr属性に格納\n        self.weight_decay = weight_decay # コンストラクタに渡された重み減衰をself.weight_decay属性に格納\n        self.validation_step_outputs = [] # 検証ステップの出力を格納するための空のリストを初期化\n\n    # NNモデルの順伝播\n    def forward(self, x):\n        return 5 * self.model(x).squeeze(-1)  # 出力は一次元テンソル\n\n    # 学習ステップ 損失関数を計算し、ログを記録\n    def training_step(self, batch):\n        x, y, w = batch\n        y_hat = self(x)\n        loss = F.mse_loss(y_hat, y, reduction='none') * w  # サンプルの重みを考慮\n        loss = loss.mean() # 損失平均\n        self.log('train_loss', loss, on_step=False, on_epoch=True, batch_size=x.size(0))\n        return loss\n\n    # 検証ステップ 損失関数を計算し、ログを記録\n    def validation_step(self, batch):\n        x, y, w = batch\n        y_hat = self(x)\n        loss = F.mse_loss(y_hat, y, reduction='none') * w\n        loss = loss.mean()\n        self.log('val_loss', loss, on_step=False, on_epoch=True, batch_size=x.size(0))\n        self.validation_step_outputs.append((y_hat, y, w))\n        return loss\n\n    # 検証エポックの最後に実行 カスタムR2スコアを計算し、ログを記録\n    def on_validation_epoch_end(self):\n        \"\"\"Calculate validation WRMSE at the end of the epoch.\"\"\"\n        y = torch.cat([x[1] for x in self.validation_step_outputs]).cpu().numpy()\n        if self.trainer.sanity_checking:\n            prob = torch.cat([x[0] for x in self.validation_step_outputs]).cpu().numpy()\n        else:\n            prob = torch.cat([x[0] for x in self.validation_step_outputs]).cpu().numpy()\n            weights = torch.cat([x[2] for x in self.validation_step_outputs]).cpu().numpy()\n            # r2_val\n            val_r_square = r2_val(y, prob, weights)\n            self.log(\"val_r_square\", val_r_square, prog_bar=True, on_step=False, on_epoch=True)\n        self.validation_step_outputs.clear()\n\n    # 最適化アルゴリズムと学習率スケジューラを設定\n    def configure_optimizers(self):\n        optimizer = torch.optim.Adam(self.parameters(), lr=self.lr, weight_decay=self.weight_decay)\n        scheduler = torch.optim.lr_scheduler.ReduceLROnPlateau(optimizer, mode='min', factor=0.5, patience=5,\n                                                               verbose=True)\n        return {\n            'optimizer': optimizer,\n            'lr_scheduler': {\n                'scheduler': scheduler,\n                'monitor': 'val_loss',\n            }\n        }\n\n    # 学習エポック終了時に、学習結果のメトリクスをログ出力\n    def on_train_epoch_end(self):\n        if self.trainer.sanity_checking:\n            return\n        epoch = self.trainer.current_epoch\n        metrics = {k: v.item() if isinstance(v, torch.Tensor) else v for k, v in self.trainer.logged_metrics.items()}\n        formatted_metrics = {k: f\"{v:.5f}\" for k, v in metrics.items()}\n        print(f\"Epoch {epoch}: {formatted_metrics}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T08:34:55.288174Z","iopub.execute_input":"2024-12-31T08:34:55.288500Z","iopub.status.idle":"2024-12-31T08:34:55.304333Z","shell.execute_reply.started":"2024-12-31T08:34:55.288471Z","shell.execute_reply":"2024-12-31T08:34:55.303648Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"N_folds = 5 # モデルの数を設定\n# Load Best Model\nmodels = [] # モデルを格納するリストを初期化\nfor fold in range(N_folds): # 各フォールドに対して繰り返し処理を行う\n    checkpoint_path = f\"{CONFIG.model_paths[0]}/nn_{fold}.model\" # モデルのチェックポイントファイルパスを作成\n    model = NN.load_from_checkpoint(checkpoint_path) # チェックポイントファイルからモデルを読み込む\n    models.append(model.to(\"cuda:0\")) # モデルをGPUに転送し、リストに追加","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T08:34:58.644763Z","iopub.execute_input":"2024-12-31T08:34:58.645099Z","iopub.status.idle":"2024-12-31T08:34:59.653404Z","shell.execute_reply.started":"2024-12-31T08:34:58.645071Z","shell.execute_reply":"2024-12-31T08:34:59.652366Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### CV Score","metadata":{}},{"cell_type":"code","source":"X_valid = valid[ xgb_feature_cols ] # 検証データの、XGBoostモデルが使う特徴量のみを抽出\ny_valid = valid[ CONFIG.target_col ] # 検証データのターゲット変数を抽出\nw_valid = valid[ \"weight\" ] # 検証データの重みを抽出\ny_pred_valid_xgb = xgb_model.predict(X_valid) # XGBで予測\nvalid_score = r2_score( y_valid, y_pred_valid_xgb, sample_weight=w_valid ) # R2スコアを計算\nvalid_score","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T08:35:01.885736Z","iopub.execute_input":"2024-12-31T08:35:01.886063Z","iopub.status.idle":"2024-12-31T08:35:04.345700Z","shell.execute_reply.started":"2024-12-31T08:35:01.886035Z","shell.execute_reply":"2024-12-31T08:35:04.344610Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_valid = valid[ CONFIG.feature_cols ] # nnモデルが使う特徴量を抽出\ny_valid = valid[ CONFIG.target_col ]\nw_valid = valid[ \"weight\" ]\nX_valid = X_valid.fillna(method = 'ffill').fillna(0) # 欠損値を前方補完で埋めた後、残りの欠損値を0埋め\nX_valid.shape, y_valid.shape, w_valid.shape","metadata":{"execution":{"iopub.status.busy":"2024-12-31T08:35:06.612536Z","iopub.execute_input":"2024-12-31T08:35:06.612882Z","iopub.status.idle":"2024-12-31T08:35:08.649401Z","shell.execute_reply.started":"2024-12-31T08:35:06.612854Z","shell.execute_reply":"2024-12-31T08:35:08.648367Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y_pred_valid_nn = np.zeros(y_valid.shape) # nnの予測結果を格納する配列を初期化\nwith torch.no_grad():\n    for model in models:\n        # 評価モードにし、予測を行い、予測結果を平均してy_pred_nnに格納\n        model.eval()\n        y_pred_valid_nn += model(torch.FloatTensor(X_valid.values).to(\"cuda:0\")).cpu().numpy() / len(models)\nvalid_score = r2_score( y_valid, y_pred_valid_nn, sample_weight=w_valid )\nvalid_score","metadata":{"execution":{"iopub.status.busy":"2024-12-31T08:35:11.306116Z","iopub.execute_input":"2024-12-31T08:35:11.306428Z","iopub.status.idle":"2024-12-31T08:35:17.923280Z","shell.execute_reply.started":"2024-12-31T08:35:11.306402Z","shell.execute_reply":"2024-12-31T08:35:17.922406Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y_pred_valid_ensemble = 0.5 * (y_pred_valid_xgb + y_pred_valid_nn) # XGBとnnの予測結果を平均したアンサンブル予測を計算\nvalid_score = r2_score( y_valid, y_pred_valid_ensemble, sample_weight=w_valid )\nvalid_score","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T08:35:23.259416Z","iopub.execute_input":"2024-12-31T08:35:23.260022Z","iopub.status.idle":"2024-12-31T08:35:23.280183Z","shell.execute_reply.started":"2024-12-31T08:35:23.259990Z","shell.execute_reply":"2024-12-31T08:35:23.279333Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"del valid, X_valid, y_valid, w_valid # 使用済みの変数を削除\ngc.collect() # 不要になったメモリを解放","metadata":{"execution":{"iopub.status.busy":"2024-12-31T08:35:25.667456Z","iopub.execute_input":"2024-12-31T08:35:25.668067Z","iopub.status.idle":"2024-12-31T08:35:25.847975Z","shell.execute_reply.started":"2024-12-31T08:35:25.668032Z","shell.execute_reply":"2024-12-31T08:35:25.847073Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"---\n# **《《《Finaly Inf&Blend》》》**\n---","metadata":{}},{"cell_type":"code","source":"lags_ : pl.DataFrame | None = None # グローバル変数としてlags_を定義\n\n# 推論を行う関数\ndef predict(test: pl.DataFrame, lags: pl.DataFrame | None) -> pl.DataFrame | pd.DataFrame:\n    global lags_\n    # lagsが与えられたら、更新\n    if lags is not None:\n        lags_ = lags\n\n    # 推論結果を格納するDFを作成\n    predictions = test.select(\n        'row_id',\n        pl.lit(0.0).alias('responder_6'),\n    )\n    # シンボルIDを抽出\n    symbol_ids = test.select('symbol_id').to_numpy()[:, 0]\n\n    # lagsのDFから、前日最後のデータを取得\n    lags = lags_.clone().group_by([\"date_id\", \"symbol_id\"], maintain_order=True).last() # pick up last record of previous date\n\n    # テストデータにラグ特徴量を結合\n    test = test.join(lags, on=[\"date_id\", \"symbol_id\"],  how=\"left\")\n\n    # ------------------------------------------------- #\n    # Inf\n    # ------------------------------------------------- #\n    preds1 = np.zeros((test.shape[0],)) # XGBの予測結果を格納する配列を初期化\n    preds2 = np.zeros((test.shape[0],)) # nnの予測結果を格納する配列を初期化\n    \n    \"\"\" Pred XGB \"\"\"\n    preds1 += xgb_model.predict(test[xgb_feature_cols].to_pandas())/4\n    # preds1 += xgb_model1.predict(test[xgb_feature_cols].to_pandas())/5\n    preds1 += xgb_model2.predict(test[xgb_feature_cols].to_pandas())/4\n    # preds1 += xgb_model3.predict(test[xgb_feature_cols].to_pandas())/5\n    preds1 += xgb_model4.predict(test[xgb_feature_cols].to_pandas())/4\n    preds1 += xgb_model5.predict(test[xgb_feature_cols].to_pandas())/4\n    \n    \"\"\" Pred NN \"\"\"\n    test_input = test[CONFIG.feature_cols].to_pandas()\n    test_input = test_input.fillna(method = 'ffill').fillna(0)\n    test_input = torch.FloatTensor(test_input.values).to(\"cuda:0\")\n    with torch.no_grad():\n        for i, nn_model in enumerate(tqdm(models)):\n            nn_model.eval()\n            preds2 += nn_model(test_input).cpu().numpy()/len(models)\n\n    \"\"\" Model Weight \"\"\"\n    _ModelW=[0.56,0.50] # モデルの重みづけを設定\n    preds = (preds1*_ModelW[0] + preds2*_ModelW[1]) # 重みづけ平均により、最終予測値を生成\n\n    \"\"\" Finaly \"\"\"\n    predictions = \\\n    test.select('row_id').\\\n    with_columns(\n        pl.Series(\n            name   = 'responder_6', \n            values = np.clip(preds, a_min = -5, a_max = 5), # -5~5の範囲でクリップし、responder_6列としてDFに追加\n            dtype  = pl.Float64,\n        )\n    )\n\n    # 予測結果の形式を確認\n    assert isinstance(predictions, pl.DataFrame | pd.DataFrame)\n    assert list(predictions.columns) == ['row_id', 'responder_6']\n    assert len(predictions) == len(test)\n\n    return predictions","metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","papermill":{"duration":0.018344,"end_time":"2024-10-10T11:58:33.59684","exception":false,"start_time":"2024-10-10T11:58:33.578496","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2024-12-31T08:35:53.654468Z","iopub.execute_input":"2024-12-31T08:35:53.655044Z","iopub.status.idle":"2024-12-31T08:35:53.665383Z","shell.execute_reply.started":"2024-12-31T08:35:53.655012Z","shell.execute_reply":"2024-12-31T08:35:53.664428Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"inference_server = kaggle_evaluation.jane_street_inference_server.JSInferenceServer(predict)\n\nif os.getenv('KAGGLE_IS_COMPETITION_RERUN'):\n    inference_server.serve()\nelse:\n    inference_server.run_local_gateway(\n        (\n            '/kaggle/input/jane-street-realtime-marketdata-forecasting/test.parquet',\n            '/kaggle/input/jane-street-realtime-marketdata-forecasting/lags.parquet',\n        )\n    )","metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","papermill":{"duration":2.225871,"end_time":"2024-10-10T11:58:35.830964","exception":false,"start_time":"2024-10-10T11:58:33.605093","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2024-12-31T08:35:57.976670Z","iopub.execute_input":"2024-12-31T08:35:57.977014Z","iopub.status.idle":"2024-12-31T08:35:58.377099Z","shell.execute_reply.started":"2024-12-31T08:35:57.976988Z","shell.execute_reply":"2024-12-31T08:35:58.374843Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}