{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":84493,"databundleVersionId":9871156,"sourceType":"competition"},{"sourceId":9756372,"sourceType":"datasetVersion","datasetId":5973876},{"sourceId":9801075,"sourceType":"datasetVersion","datasetId":6006872},{"sourceId":9806342,"sourceType":"datasetVersion","datasetId":6010899},{"sourceId":203900450,"sourceType":"kernelVersion"}],"dockerImageVersionId":30823,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport torch\nfrom torch.utils.data import DataLoader, TensorDataset\nimport gc\nfrom sklearn.metrics import r2_score\nfrom tqdm import tqdm\n\ndef preprocess_data(data: pd.DataFrame, feature_cols: list) -> torch.FloatTensor:\n    \"\"\"Prepare input data by filling missing values.\"\"\"\n    data = data[feature_cols].fillna(method='ffill').fillna(0)\n    return torch.FloatTensor(data.values)\n\ndef predict_with_models(models, input_data: torch.FloatTensor) -> np.ndarray:\n    \"\"\"Predict using multiple PyTorch models.\"\"\"\n    predictions = np.zeros((input_data.shape[0],))\n    with torch.no_grad():\n        for model in models:\n            model.eval()\n            predictions += model(input_data).cpu().numpy() / len(models)\n    return predictions\n\ndef clear_memory(*variables):\n    \"\"\"Clear variables from memory and force garbage collection.\"\"\"\n    for var in variables:\n        del var\n    gc.collect()\n\nimport polars as pl\nimport os, gc\nfrom matplotlib import pyplot as plt\nimport pickle\n\nimport torch\nimport torch.nn as nn\nimport torch.nn.functional as F\nfrom pytorch_lightning import (LightningModule)\nfrom pytorch_lightning.callbacks import EarlyStopping, ModelCheckpoint, Timer\n\nfrom sklearn.model_selection import train_test_split\nfrom torch.utils.data import Dataset, DataLoader\nfrom lightgbm import LGBMRegressor\nimport lightgbm as lgb\nfrom xgboost import XGBRegressor\nfrom catboost import CatBoostRegressor\nfrom sklearn.ensemble import VotingRegressor\nfrom sklearn.linear_model import Ridge\nimport warnings\nwarnings.filterwarnings('ignore')\npd.options.display.max_columns = None\n\nimport kaggle_evaluation.jane_street_inference_server","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T04:43:53.322361Z","iopub.execute_input":"2024-12-22T04:43:53.322683Z","iopub.status.idle":"2024-12-22T04:43:53.330197Z","shell.execute_reply.started":"2024-12-22T04:43:53.322662Z","shell.execute_reply":"2024-12-22T04:43:53.329212Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class CONFIG:\n    seed = 42\n    target_col = \"responder_6\"\n    feature_cols = [f\"feature_{idx:02d}\" for idx in range(79)] + [f\"responder_{idx}_lag_1\" for idx in range(9)]\n    nn_model_dir = \"/kaggle/input/js-xs-nn-trained-model\"\n    xgb_model_path = \"/kaggle/input/js-with-lags-trained-xgb/result.pkl\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T04:41:09.676706Z","iopub.execute_input":"2024-12-22T04:41:09.677098Z","iopub.status.idle":"2024-12-22T04:41:09.682538Z","shell.execute_reply.started":"2024-12-22T04:41:09.677065Z","shell.execute_reply":"2024-12-22T04:41:09.681612Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def clear_memory(*variables):\n    for var in variables:\n        del var\n    gc.collect()\n\ndef preprocess_data(data: pd.DataFrame, feature_cols: list) -> torch.FloatTensor:\n    data = data[feature_cols].fillna(method='ffill').fillna(0)\n    return torch.FloatTensor(data.values)\n\ndef r2_val(y_true, y_pred, sample_weight):\n    r2 = 1 - np.average((y_pred - y_true) ** 2, weights=sample_weight) / (np.average((y_true) ** 2, weights=sample_weight) + 1e-41)\n    return r2","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T04:41:10.084411Z","iopub.execute_input":"2024-12-22T04:41:10.084685Z","iopub.status.idle":"2024-12-22T04:41:10.089920Z","shell.execute_reply.started":"2024-12-22T04:41:10.084664Z","shell.execute_reply":"2024-12-22T04:41:10.088966Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 一部特徴量に対してローリング平均などを付加する関数を用意。\ndef add_rolling_features(df: pd.DataFrame, group_cols=[\"symbol_id\"], roll_window=5):\n    # symbolごとにrolling mean\n    # 本例では非常に簡易的な例\n    df = df.sort_values([\"symbol_id\",\"time_id\"])  # 時系列ソート\n    for f in CONFIG.feature_cols:\n        new_col = f\"{f}_rollmean_{roll_window}\"\n        df[new_col] = df.groupby(group_cols)[f].rolling(roll_window, min_periods=1).mean().values\n    return df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T04:41:10.568087Z","iopub.execute_input":"2024-12-22T04:41:10.568388Z","iopub.status.idle":"2024-12-22T04:41:10.572869Z","shell.execute_reply.started":"2024-12-22T04:41:10.568356Z","shell.execute_reply":"2024-12-22T04:41:10.571940Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class NN(LightningModule):\n    def __init__(self, input_dim, hidden_dims, dropouts, lr, weight_decay):\n        super().__init__()\n        self.save_hyperparameters()\n        layers = []\n        in_dim = input_dim\n        for i, hidden_dim in enumerate(hidden_dims):\n            layers.append(nn.BatchNorm1d(in_dim))\n            if i > 0:\n                layers.append(nn.SiLU())\n            if i < len(dropouts):\n                layers.append(nn.Dropout(dropouts[i]))\n            layers.append(nn.Linear(in_dim, hidden_dim))\n            in_dim = hidden_dim\n        layers.append(nn.Linear(in_dim, 1))\n        layers.append(nn.Tanh())\n        self.model = nn.Sequential(*layers)\n        self.lr = lr\n        self.weight_decay = weight_decay\n        self.validation_step_outputs = []\n\n    def forward(self, x):\n        return 5 * self.model(x).squeeze(-1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T04:41:14.207427Z","iopub.execute_input":"2024-12-22T04:41:14.207706Z","iopub.status.idle":"2024-12-22T04:41:14.213840Z","shell.execute_reply.started":"2024-12-22T04:41:14.207684Z","shell.execute_reply":"2024-12-22T04:41:14.212711Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"valid = pl.scan_parquet(\n    f\"/kaggle/input/js24-preprocessing-create-lags/validation.parquet/\"\n).collect().to_pandas()\n\n# 追加特徴量例\nvalid = add_rolling_features(valid)\n\nXGB_feature_cols = [\"symbol_id\", \"time_id\"] + CONFIG.feature_cols\n\n# XGBモデルロード\nwith open(CONFIG.xgb_model_path, \"rb\") as fp:\n    result = pickle.load(fp)\n    xgb_model = result[\"model\"]\n\n# NNモデルロード\nN_folds = 5  # foldを1などに制限して高速化する例\nmodels_nn = []\ncheckpoint_path = f\"{CONFIG.nn_model_dir}/nn_0.model\"\nmodel = NN.load_from_checkpoint(checkpoint_path)\nmodels_nn.append(model.to(\"cuda:0\"))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T04:44:36.696704Z","iopub.execute_input":"2024-12-22T04:44:36.696991Z","iopub.status.idle":"2024-12-22T04:45:02.282338Z","shell.execute_reply.started":"2024-12-22T04:44:36.696970Z","shell.execute_reply":"2024-12-22T04:45:02.281671Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Validationデータ用意\nX_valid_df = valid.copy()\nX_valid = preprocess_data(X_valid_df, CONFIG.feature_cols)\ny_valid = valid[CONFIG.target_col].values\nw_valid = valid[\"weight\"].values","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T04:45:02.283275Z","iopub.execute_input":"2024-12-22T04:45:02.283509Z","iopub.status.idle":"2024-12-22T04:45:06.291784Z","shell.execute_reply.started":"2024-12-22T04:45:02.283488Z","shell.execute_reply":"2024-12-22T04:45:06.291055Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# NN予測\ndef predict_with_nn(models, input_data: torch.FloatTensor):\n    preds = np.zeros((input_data.shape[0],))\n    with torch.no_grad():\n        for m in models:\n            m.eval()\n            preds += m(input_data).cpu().numpy() / len(models)\n    return preds\n\ny_pred_valid_nn = predict_with_nn(models_nn, X_valid.to(\"cuda:0\"))\nvalid_score_nn = r2_score(y_valid, y_pred_valid_nn, sample_weight=w_valid)\nprint(f\"Validation R2 score (NN): {valid_score_nn:.5f}\")\n\n# XGB予測\nX_valid_xgb = X_valid_df[XGB_feature_cols].fillna(method='ffill').fillna(0)\ny_pred_valid_xgb = xgb_model.predict(X_valid_xgb)\nvalid_score_xgb = r2_score(y_valid, y_pred_valid_xgb, sample_weight=w_valid)\nprint(f\"Validation R2 score (XGB): {valid_score_xgb:.5f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T04:45:06.292928Z","iopub.execute_input":"2024-12-22T04:45:06.293173Z","iopub.status.idle":"2024-12-22T04:45:12.073781Z","shell.execute_reply.started":"2024-12-22T04:45:06.293144Z","shell.execute_reply":"2024-12-22T04:45:12.072943Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"lgb_model = LGBMRegressor(n_estimators=500, learning_rate=0.05, random_state=CONFIG.seed)\nlgb_model.fit(X_valid_xgb, y_valid, sample_weight=w_valid)\ny_pred_valid_lgb = lgb_model.predict(X_valid_xgb)\nvalid_score_lgb = r2_score(y_valid, y_pred_valid_lgb, sample_weight=w_valid)\nprint(f\"Validation R2 score (LGB): {valid_score_lgb:.5f}\")\n\ncat_model = CatBoostRegressor(iterations=500, learning_rate=0.05, depth=6, verbose=0, random_seed=CONFIG.seed)\ncat_model.fit(X_valid_xgb, y_valid, sample_weight=w_valid)\ny_pred_valid_cat = cat_model.predict(X_valid_xgb)\nvalid_score_cat = r2_score(y_valid, y_pred_valid_cat, sample_weight=w_valid)\nprint(f\"Validation R2 score (CAT): {valid_score_cat:.5f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T04:45:12.485364Z","iopub.execute_input":"2024-12-22T04:45:12.485643Z","iopub.status.idle":"2024-12-22T04:47:58.650681Z","shell.execute_reply.started":"2024-12-22T04:45:12.485620Z","shell.execute_reply":"2024-12-22T04:47:58.649691Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# スタッキング\nstack_features = np.vstack([\n    y_pred_valid_xgb,\n    y_pred_valid_nn,\n    y_pred_valid_lgb,\n    y_pred_valid_cat\n]).T\n\nmeta_model = Ridge(alpha=1.0)\nmeta_model.fit(stack_features, y_valid, sample_weight=w_valid)\ny_pred_valid_ensemble = meta_model.predict(stack_features)\nvalid_score_ensemble = r2_score(y_valid, y_pred_valid_ensemble, sample_weight=w_valid)\nprint(f\"Validation R2 score (Ensemble): {valid_score_ensemble:.5f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T04:47:58.651906Z","iopub.execute_input":"2024-12-22T04:47:58.652243Z","iopub.status.idle":"2024-12-22T04:47:58.799055Z","shell.execute_reply.started":"2024-12-22T04:47:58.652213Z","shell.execute_reply":"2024-12-22T04:47:58.796539Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 予測\nlags_ : pl.DataFrame | None = None\n\ndef predict(test: pl.DataFrame, lags: pl.DataFrame | None) -> pl.DataFrame | pd.DataFrame:\n    global lags_\n    if lags is not None:\n        lags_ = lags\n\n    # lag特徴量付与\n    if lags is not None:\n        lags = lags.group_by([\"date_id\", \"symbol_id\"], maintain_order=True).last() \n        test = test.join(lags, on=[\"date_id\", \"symbol_id\"], how=\"left\")\n    else:\n        for idx in range(9):\n            test = test.with_columns(\n                pl.lit(0.0).alias(f'responder_{idx}_lag_1')\n            )\n\n    # 新たなrolling特徴量を付加\n    # 注意: 本番でtest全体にrollingはできないため、シンプルな例示\n    # 実際にはtime_id順でテストのスライド処理が必要、ここでは簡略化\n    test_pd = test.to_pandas()\n    #test_pd = add_rolling_features(test_pd) # 推論時間を減らすためコメントアウト\n\n    # 前処理\n    test_xgb = test_pd[XGB_feature_cols].fillna(method='ffill').fillna(0)\n\n    # 個別モデル予測\n    preds_xgb = xgb_model.predict(test_xgb)\n    test_nn_input = preprocess_data(test_pd, CONFIG.feature_cols)\n    test_nn_input = test_nn_input.to(\"cuda:0\")\n    preds_nn = predict_with_nn(models_nn, test_nn_input)\n    preds_lgb = lgb_model.predict(test_xgb)\n    preds_cat = cat_model.predict(test_xgb)\n\n    # Ensemble\n    stack_test = np.vstack([preds_xgb, preds_nn, preds_lgb, preds_cat]).T\n    final_preds = meta_model.predict(stack_test)\n\n    predictions = (\n        test.select('row_id')\n        .with_columns(\n            pl.Series(\n                name='responder_6',\n                values=np.clip(final_preds, a_min=-5, a_max=5),\n                dtype=pl.Float64,\n            )\n        )\n    )\n\n    # Check\n    assert isinstance(predictions, (pl.DataFrame, pd.DataFrame))\n    assert list(predictions.columns) == ['row_id', 'responder_6']\n    assert len(predictions) == len(test)\n\n    return predictions","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T04:47:58.800815Z","iopub.execute_input":"2024-12-22T04:47:58.801200Z","iopub.status.idle":"2024-12-22T04:47:58.821672Z","shell.execute_reply.started":"2024-12-22T04:47:58.801163Z","shell.execute_reply":"2024-12-22T04:47:58.820541Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"inference_server = kaggle_evaluation.jane_street_inference_server.JSInferenceServer(predict)\n\nif os.getenv('KAGGLE_IS_COMPETITION_RERUN'):\n    inference_server.serve()\nelse:\n    inference_server.run_local_gateway(\n        (\n            '/kaggle/input/jane-street-real-time-market-data-forecasting/test.parquet',\n            '/kaggle/input/jane-street-real-time-market-data-forecasting/lags.parquet',\n        )\n    )","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T04:47:58.822816Z","iopub.execute_input":"2024-12-22T04:47:58.823195Z","iopub.status.idle":"2024-12-22T04:47:59.052188Z","shell.execute_reply.started":"2024-12-22T04:47:58.823157Z","shell.execute_reply":"2024-12-22T04:47:59.051241Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}