{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":84493,"databundleVersionId":9871156,"sourceType":"competition"},{"sourceId":9970777,"sourceType":"datasetVersion","datasetId":6134157},{"sourceId":9971012,"sourceType":"datasetVersion","datasetId":6134332}],"dockerImageVersionId":30787,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"- [Preorocessiong](https://www.kaggle.com/code/grooper/js24-preprocessing-create-lags-forked)\n- [Baseline](https://www.kaggle.com/code/mmiyashita/js24-baseline)","metadata":{}},{"cell_type":"markdown","source":"# Libraries","metadata":{}},{"cell_type":"code","source":"import os\nfrom tqdm.auto import tqdm\nimport numpy as np\nimport polars as pl\nimport pandas as pd\nimport matplotlib.pyplot as plt\nfrom sklearn.metrics import mean_squared_error\n\nimport kaggle_evaluation.jane_street_inference_server","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-05T07:11:56.013719Z","iopub.execute_input":"2024-12-05T07:11:56.014243Z","iopub.status.idle":"2024-12-05T07:11:59.690274Z","shell.execute_reply.started":"2024-12-05T07:11:56.014194Z","shell.execute_reply":"2024-12-05T07:11:59.688809Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Configurations","metadata":{}},{"cell_type":"code","source":"class CONFIG:\n    seed = 42\n    target_col = \"responder_6\"\n    feature_cols = (\n        ['date_id', 'time_id', 'symbol_id']\n        + [f\"feature_{i:02}\" for i in range(79)]\n        # + [f\"responder_{i}\" for i in range(9)]\n        + [f\"responder_{i}_lag_1\" for i in range(9)]\n    )\n    weight_col = \"weight\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T07:14:04.210984Z","iopub.execute_input":"2024-12-05T07:14:04.211715Z","iopub.status.idle":"2024-12-05T07:14:04.218666Z","shell.execute_reply.started":"2024-12-05T07:14:04.211672Z","shell.execute_reply":"2024-12-05T07:14:04.217192Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Load Data","metadata":{}},{"cell_type":"code","source":"train = pl.scan_parquet(\"/kaggle/input/train-data-js24-20241121/training_with_lags_not_partition.parquet\")\nvalid = pl.scan_parquet(\"/kaggle/input/valid-data-js24-20241121/validation_with_lags_not_partition.parquet\")\n\n# LazyFrame -> DataFrame\n# .sample(fraction=0.1, seed=42)\ntrain = train.collect().to_pandas()\nvalid = valid.collect().to_pandas()\n# train = train.collect().drop_nulls().to_pandas()\n# valid = valid.collect().drop_nulls().to_pandas()\nprint(train.shape, valid.shape)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# train\nX_train = train[CONFIG.feature_cols]\ny_train = train[CONFIG.target_col]\nw_train = train[CONFIG.weight_col]\nprint(\"train shape\")\nprint(f\"X_train: {X_train.shape}, y_train: {y_train.shape}, w_train: {w_train.shape}\")\n\n# valid\nX_valid = valid[CONFIG.feature_cols]\ny_valid = valid[CONFIG.target_col]\nw_valid = valid[CONFIG.weight_col]\nprint(\"valid shape\")\nprint(f\"X_valid: {X_valid.shape}, y_valid: {y_valid.shape}, w_valid: {w_valid.shape}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T07:16:17.038734Z","iopub.execute_input":"2024-12-05T07:16:17.039187Z","iopub.status.idle":"2024-12-05T07:16:17.725912Z","shell.execute_reply.started":"2024-12-05T07:16:17.039153Z","shell.execute_reply":"2024-12-05T07:16:17.724756Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Model testing","metadata":{}},{"cell_type":"code","source":"# Model import\nfrom xgboost import XGBRegressor\nfrom lightgbm import LGBMRegressor\nfrom sklearn.ensemble import RandomForestRegressor\nfrom sklearn.linear_model import LinearRegression\nfrom sklearn.svm import SVR\nfrom sklearn.neighbors import KNeighborsRegressor\nfrom catboost import CatBoostRegressor, Pool\nfrom sklearn.linear_model import Ridge, Lasso\nfrom sklearn.tree import DecisionTreeRegressor\nfrom sklearn.ensemble import GradientBoostingRegressor\nfrom sklearn.ensemble import HistGradientBoostingRegressor\nfrom sklearn.neural_network import MLPRegressor","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T07:16:20.074915Z","iopub.execute_input":"2024-12-05T07:16:20.075290Z","iopub.status.idle":"2024-12-05T07:16:21.885853Z","shell.execute_reply.started":"2024-12-05T07:16:20.075260Z","shell.execute_reply":"2024-12-05T07:16:21.884821Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# eval metric\n# for eval\ndef custom_r2(y_true, y_pred, sample_weight):\n    y_true = np.array(y_true)\n    y_pred = np.array(y_pred)\n    if sample_weight is not None:\n        sample_weight = np.array(sample_weight)\n    else:\n        sample_weight = np.ones_like(y_true)\n    r2 = 1 - np.sum(sample_weight * (y_true - y_pred) ** 2 ) / (np.sum(sample_weight * y_true ** 2, ))\n    return r2\n\n\n# Custom R2 metric for LightGBM\ndef r2_lgb(y_true, y_pred, sample_weight):\n    r2 = 1 - np.average((y_pred - y_true) ** 2, weights=sample_weight) / (np.average((y_true) ** 2, weights=sample_weight) + 1e-38)\n    return 'r2', r2, True\n    \n# Custom R2 metric for XGBoost\ndef r2_xgb(y_true, y_pred, sample_weight):\n    r2 = 1 - np.sum(sample_weight * (y_true - y_pred) ** 2 ) / (np.sum(sample_weight * y_true ** 2, ))\n    return -r2\n\n# Custom R2 metric for CatBoost\nclass r2_cbt(object):\n    def get_final_error(self, error, weight):\n        return 1 - error / (weight + 1e-38)\n\n    def is_max_optimal(self):\n        return True\n\n    def evaluate(self, approxes, target, weight):\n        assert len(approxes) == 1\n        assert len(target) == len(approxes[0])\n\n        approx = approxes[0]\n\n        error_sum = 0.0\n        weight_sum = 0.0\n\n        for i in range(len(approx)):\n            w = 1.0 if weight is None else weight[i]\n            weight_sum += w * (target[i] ** 2)\n            error_sum += w * ((approx[i] - target[i]) ** 2)\n\n        return error_sum, weight_sum","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T07:22:40.752655Z","iopub.execute_input":"2024-12-05T07:22:40.753141Z","iopub.status.idle":"2024-12-05T07:22:40.763823Z","shell.execute_reply.started":"2024-12-05T07:22:40.753105Z","shell.execute_reply":"2024-12-05T07:22:40.762690Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# eval function\ndef evaluate_model(model, X_train, y_train, X_valid, y_valid, w_train=None, w_valid=None):\n    if model_name == 'LightGBM':\n        # Train LightGBM model with early stopping and evaluation logging\n        model.fit(X_train, y_train, w_train,  \n                  eval_metric=[r2_lgb],\n                  eval_set=[(X_valid, y_valid, w_valid)],)\n    elif model_name == 'CatBoost':\n        # Prepare evaluation set for CatBoost\n        evalset = Pool(X_valid, y_valid, weight=w_valid)\n        \n        # Train CatBoost model with early stopping and verbose logging\n        model.fit(X_train, y_train, sample_weight=w_train, \n                  eval_set=[evalset], \n                  verbose=100, )\n    elif model_name == 'XGBoost':\n        model.fit(X_train, y_train, sample_weight=w_train, \n                  eval_set=[(X_valid, y_valid)], \n                  sample_weight_eval_set=[w_valid], \n                  verbose=200, )\n    else:\n        model.fit(X_train, y_train)\n        \n    y_pred = model.predict(X_valid)\n    \n    r2 = custom_r2(y_valid, y_pred, sample_weight=w_valid)\n    mse = mean_squared_error(y_valid, y_pred, sample_weight=w_valid)\n    \n    plt.figure(figsize=(5, 3))\n    plt.plot(y_valid[:100], label=\"True Values\", linestyle=\"--\", color=\"black\", linewidth=2)\n    plt.plot(y_pred[:100], label=f\"Predictions ({model.__class__.__name__})\", alpha=0.7)\n    plt.title(f\"{model.__class__.__name__} Predictions vs True Values (R²: {r2:.2f})\")\n    plt.xlabel(\"Sample Index\")\n    plt.ylabel(\"Prediction Value\")\n    plt.legend()\n    plt.grid(True)\n    plt.show()\n    \n    return {\"model\": model, \"mse\": mse, \"r2\": r2, \"predictions\": y_pred}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T07:22:54.238232Z","iopub.execute_input":"2024-12-05T07:22:54.238658Z","iopub.status.idle":"2024-12-05T07:22:54.250664Z","shell.execute_reply.started":"2024-12-05T07:22:54.238622Z","shell.execute_reply":"2024-12-05T07:22:54.249245Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# model dict can use with nan\nmodels = {\n    \"XGBoost\": XGBRegressor(\n        n_estimators=3000,\n        max_depth=6,\n        learning_rate=0.01,\n        tree_method=\"hist\",\n        objective=\"reg:squarederror\",\n        eval_metric=r2_xgb,\n        disable_default_eval_metric=True,\n        device=\"cuda\"\n    ),\n    \"LightGBM\": LGBMRegressor(\n        n_estimators=500,\n        max_depth=11,\n        learning_rate=0.005,\n        num_leaves=100\n    ),\n    \"CatBoost\": CatBoostRegressor(iterations=1000, learning_rate=0.05, task_type='GPU', loss_function='MAPE', eval_metric=r2_cbt()),\n    \"GaussianProcess\": HistGradientBoostingRegressor(),\n}\n# model can't use with nan\n# models = {\n#     \"LinearRegression\": LinearRegression(),\n#     \"Ridge\": Ridge(alpha=1.0),\n#     \"Lasso\": Lasso(alpha=0.1),\n#     # \"DecisionTree\": DecisionTreeRegressor(max_depth=6),\n#     # \"GradientBoosting\": GradientBoostingRegressor(\n#     #     n_estimators=100, \n#     #     learning_rate=0.01,\n#     #     max_depth=6\n#     # ),\n#     # \"MLP\": MLPRegressor(hidden_layer_sizes=(50, 50), max_iter=1000),\n#     # \"RandomForest\": RandomForestRegressor(\n#     #     n_estimators=10,\n#     #     max_depth=6,\n#     #     random_state=42\n#     # ),\n# }","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T07:22:46.028806Z","iopub.execute_input":"2024-12-05T07:22:46.029166Z","iopub.status.idle":"2024-12-05T07:22:46.035300Z","shell.execute_reply.started":"2024-12-05T07:22:46.029137Z","shell.execute_reply":"2024-12-05T07:22:46.034169Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# evaluate\nresults = {}\n\n# Train and evaluate each model\nfor model_name, model in models.items():\n    print(f\"Training {model_name}...\")\n    result = evaluate_model(model, X_train, y_train, X_valid, y_valid, w_train, w_valid)\n    results[model_name] = result\n    print(f\"{model_name} - MSE: {result['mse']:.4f}, R2: {result['r2']:.4f}\")\n\n# Find the best model\nbest_model_name = max(results, key=lambda name: abs(results[name][\"r2\"]))\nprint(f\"Best model: {best_model_name} with R2: {results[best_model_name]['r2']:.4f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T07:22:57.571571Z","iopub.execute_input":"2024-12-05T07:22:57.571997Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# # evaluate\n# results = {}\n\n# # Train and evaluate each model\n# for model_name, model in models.items():\n#     print(f\"Training {model_name}...\")\n#     result = evaluate_model(model, X_train, y_train, X_valid, y_valid, w_train, w_valid)\n#     results[model_name] = result\n#     print(f\"{model_name} - MSE: {result['mse']:.4f}, R2: {result['r2']:.4f}\")\n\n# # Find the best model\n# best_model_name = max(results, key=lambda name: abs(results[name][\"r2\"]))\n# print(f\"Best model: {best_model_name} with R2: {results[best_model_name]['r2']:.4f}\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"lags_ : pl.DataFrame | None = None\n\n\n# Replace this function with your inference code.\n# You can return either a Pandas or Polars dataframe, though Polars is recommended.\n# Each batch of predictions (except the very first) must be returned within 1 minute of the batch features being provided.\ndef predict(test: pl.DataFrame, lags: pl.DataFrame | None) -> pl.DataFrame | pd.DataFrame:\n    \"\"\"Make a prediction.\"\"\"\n    # All the responders from the previous day are passed in at time_id == 0. We save them in a global variable for access at every time_id.\n    # Use them as extra features, if you like.\n    global lags_\n    if lags is not None:\n        lags_ = lags\n\n    # Replace this section with your own predictions\n    predictions = test.select(\n        'row_id',\n        pl.lit(0.0).alias('responder_6'),\n    )\n    symbol_ids = test.select('symbol_id').to_numpy()[:, 0]\n    if not lags is None:\n        lags = lags.group_by([\"date_id\", \"symbol_id\"], maintain_order=True).last() # pick up last record of previous date\n        test = test.join(lags, on=[\"date_id\", \"symbol_id\"],  how=\"left\")\n    else:\n        test = test.with_columns(\n            ( pl.lit(0.0).alias(f'responder_{idx}_lag_1') for idx in range(9) )\n        )\n    \n    preds = np.zeros((test.shape[0],))\n    for model_name, model in models.items():\n        preds += model.predict(test[CONFIG.feature_cols].to_pandas()) / len(models)\n    print(f\"predict> preds.shape =\", preds.shape)\n\n    predictions = \\\n    test.select('row_id').\\\n    with_columns(\n        pl.Series(\n            name   = 'responder_6', \n            values = np.clip(preds, a_min = -5, a_max = 5),\n            dtype  = pl.Float64,\n        )\n    )\n\n    if isinstance(predictions, pl.DataFrame):\n        assert predictions.columns == ['row_id', 'responder_6']\n    elif isinstance(predictions, pd.DataFrame):\n        assert (predictions.columns == ['row_id', 'responder_6']).all()\n    else:\n        raise TypeError('The predict function must return a DataFrame')\n    # Confirm has as many rows as the test data.\n    assert len(predictions) == len(test)\n\n    return predictions","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T04:08:12.492524Z","iopub.execute_input":"2024-12-05T04:08:12.492887Z","iopub.status.idle":"2024-12-05T04:08:12.505954Z","shell.execute_reply.started":"2024-12-05T04:08:12.492856Z","shell.execute_reply":"2024-12-05T04:08:12.504983Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"inference_server = kaggle_evaluation.jane_street_inference_server.JSInferenceServer(predict)\n\nif os.getenv('KAGGLE_IS_COMPETITION_RERUN'):\n    inference_server.serve()\nelse:\n    inference_server.run_local_gateway(\n        (\n            '/kaggle/input/jane-street-real-time-market-data-forecasting/test.parquet',\n            '/kaggle/input/jane-street-real-time-market-data-forecasting/lags.parquet',\n        )\n    )","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T04:08:16.126705Z","iopub.execute_input":"2024-12-05T04:08:16.127461Z","iopub.status.idle":"2024-12-05T04:08:16.338950Z","shell.execute_reply.started":"2024-12-05T04:08:16.127425Z","shell.execute_reply":"2024-12-05T04:08:16.337988Z"}},"outputs":[],"execution_count":null}]}