{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"codemirror_mode":{"name":"ipython","version":3},"file_extension":".py","mimetype":"text/x-python","name":"python","nbconvert_exporter":"python","pygments_lexer":"ipython3","version":"3.10.14"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":84493,"databundleVersionId":9871156,"sourceType":"competition"},{"sourceId":9756372,"sourceType":"datasetVersion","datasetId":5973876},{"sourceId":203900450,"sourceType":"kernelVersion"}],"dockerImageVersionId":30787,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true},"papermill":{"default_parameters":{},"duration":7.594014,"end_time":"2024-10-10T11:58:36.355301","environment_variables":{},"exception":null,"input_path":"__notebook__.ipynb","output_path":"__notebook__.ipynb","parameters":{},"start_time":"2024-10-10T11:58:28.761287","version":"2.6.0"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Baseline notebooks:\n\n- Preprocessing : https://www.kaggle.com/code/motono0223/js24-preprocessing-create-lags\n- Training (Code only) : https://www.kaggle.com/code/motono0223/js24-train-gbdt-model-with-lags-singlemodel\n  - trained model : https://www.kaggle.com/datasets/motono0223/js24-trained-gbdt-model\n- Inference : **this notebook**  https://www.kaggle.com/code/motono0223/js24-inference-gbdt-with-lags-singlemodel\n- EDA(1) : https://www.kaggle.com/code/motono0223/eda-jane-street-real-time-market-data-forecasting\n- EDA(2) : https://www.kaggle.com/code/motono0223/eda-v2-jane-street-real-time-market-forecasting","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport polars as pl\nimport numpy as np\nimport os, gc\nfrom tqdm.auto import tqdm\nfrom matplotlib import pyplot as plt\nimport pickle\n\nfrom sklearn.metrics import r2_score\nfrom lightgbm import LGBMRegressor\nimport lightgbm as lgb\nfrom xgboost import XGBRegressor\nfrom catboost import CatBoostRegressor\nfrom sklearn.ensemble import VotingRegressor\n\nimport warnings\nwarnings.filterwarnings('ignore')\npd.options.display.max_columns = None\n\nimport kaggle_evaluation.jane_street_inference_server","metadata":{"execution":{"iopub.status.busy":"2024-11-04T19:58:55.602139Z","iopub.execute_input":"2024-11-04T19:58:55.602560Z","iopub.status.idle":"2024-11-04T19:58:55.609243Z","shell.execute_reply.started":"2024-11-04T19:58:55.602521Z","shell.execute_reply":"2024-11-04T19:58:55.608243Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Configurations","metadata":{}},{"cell_type":"code","source":"class CONFIG:\n    seed = 42\n    target_col = \"responder_6\"\n    feature_cols = [\"symbol_id\", \"time_id\"] + [f\"feature_{idx:02d}\" for idx in range(79)]+ [f\"responder_{idx}_lag_1\" for idx in range(9)]\n    \n    model_paths = [\n        #\"/kaggle/input/js24-train-gbdt-model-with-lags-singlemodel/result.pkl\",\n        \"/kaggle/input/js24-trained-gbdt-model/result.pkl\",\n    ]","metadata":{"execution":{"iopub.status.busy":"2024-11-04T19:58:55.611000Z","iopub.execute_input":"2024-11-04T19:58:55.611392Z","iopub.status.idle":"2024-11-04T19:58:55.617661Z","shell.execute_reply.started":"2024-11-04T19:58:55.611360Z","shell.execute_reply":"2024-11-04T19:58:55.616689Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load preprocessed data (to calculate CV)","metadata":{}},{"cell_type":"code","source":"valid = pl.scan_parquet(\n    f\"/kaggle/input/js24-preprocessing-create-lags/validation.parquet/\"\n).collect().to_pandas()","metadata":{"execution":{"iopub.status.busy":"2024-11-04T19:58:55.618927Z","iopub.execute_input":"2024-11-04T19:58:55.619302Z","iopub.status.idle":"2024-11-04T19:58:56.505007Z","shell.execute_reply.started":"2024-11-04T19:58:55.619259Z","shell.execute_reply":"2024-11-04T19:58:56.503963Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load model","metadata":{}},{"cell_type":"code","source":"models = []\nfor model_path in CONFIG.model_paths:\n    with open( model_path, \"rb\") as fp:\n        result = pickle.load(fp)\n\n    model = result[\"model\"]\n    models.append(model)\n\n# Show model\nfor model in models:\n    display(model)","metadata":{"execution":{"iopub.status.busy":"2024-11-04T19:58:56.507486Z","iopub.execute_input":"2024-11-04T19:58:56.508135Z","iopub.status.idle":"2024-11-04T19:58:56.528485Z","shell.execute_reply.started":"2024-11-04T19:58:56.508089Z","shell.execute_reply":"2024-11-04T19:58:56.527515Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# CV Score","metadata":{}},{"cell_type":"code","source":"X_valid = valid[ CONFIG.feature_cols ]\ny_valid = valid[ CONFIG.target_col ]\nw_valid = valid[ \"weight\" ]\n\nX_valid.shape, y_valid.shape, w_valid.shape","metadata":{"execution":{"iopub.status.busy":"2024-11-04T19:58:56.529559Z","iopub.execute_input":"2024-11-04T19:58:56.529831Z","iopub.status.idle":"2024-11-04T19:58:56.657716Z","shell.execute_reply.started":"2024-11-04T19:58:56.529801Z","shell.execute_reply":"2024-11-04T19:58:56.656485Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred_valid = model.predict(X_valid)\nvalid_score = r2_score( y_valid, y_pred_valid, sample_weight=w_valid )\nvalid_score","metadata":{"execution":{"iopub.status.busy":"2024-11-04T19:58:56.658822Z","iopub.execute_input":"2024-11-04T19:58:56.659197Z","iopub.status.idle":"2024-11-04T19:58:58.797063Z","shell.execute_reply.started":"2024-11-04T19:58:56.659160Z","shell.execute_reply":"2024-11-04T19:58:58.796039Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del valid, X_valid, y_valid, w_valid\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2024-11-04T19:58:58.798295Z","iopub.execute_input":"2024-11-04T19:58:58.798871Z","iopub.status.idle":"2024-11-04T19:58:58.950753Z","shell.execute_reply.started":"2024-11-04T19:58:58.798831Z","shell.execute_reply":"2024-11-04T19:58:58.949818Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### There seems to be bug in official code, can only submit polars dataframe","metadata":{}},{"cell_type":"code","source":"lags_ : pl.DataFrame | None = None\n    \ndef predict(test: pl.DataFrame, lags: pl.DataFrame | None) -> pl.DataFrame | pd.DataFrame:\n    global lags_\n    if lags is not None:\n        lags_ = lags\n\n    predictions = test.select(\n        'row_id',\n        pl.lit(0.0).alias('responder_6'),\n    )\n    symbol_ids = test.select('symbol_id').to_numpy()[:, 0]\n\n    if not lags is None:\n        lags = lags.group_by([\"date_id\", \"symbol_id\"], maintain_order=True).last() # pick up last record of previous date\n        test = test.join(lags, on=[\"date_id\", \"symbol_id\"],  how=\"left\")\n    else:\n        test = test.with_columns(\n            ( pl.lit(0.0).alias(f'responder_{idx}_lag_1') for idx in range(9) )\n        )\n    \n    preds = np.zeros((test.shape[0],))\n    for i, model in enumerate(tqdm(models)):\n        preds += model.predict(test[CONFIG.feature_cols].to_pandas()) / len(models)\n    print(f\"predict> preds.shape =\", preds.shape)\n    \n    predictions = \\\n    test.select('row_id').\\\n    with_columns(\n        pl.Series(\n            name   = 'responder_6', \n            values = np.clip(preds, a_min = -5, a_max = 5),\n            dtype  = pl.Float64,\n        )\n    )\n\n    # The predict function must return a DataFrame\n    assert isinstance(predictions, pl.DataFrame | pd.DataFrame)\n    # with columns 'row_id', 'responer_6'\n    assert list(predictions.columns) == ['row_id', 'responder_6']\n    # and as many rows as the test data.\n    assert len(predictions) == len(test)\n\n    return predictions","metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","papermill":{"duration":0.018344,"end_time":"2024-10-10T11:58:33.59684","exception":false,"start_time":"2024-10-10T11:58:33.578496","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2024-11-04T19:58:58.951949Z","iopub.execute_input":"2024-11-04T19:58:58.952297Z","iopub.status.idle":"2024-11-04T19:58:58.967953Z","shell.execute_reply.started":"2024-11-04T19:58:58.952261Z","shell.execute_reply":"2024-11-04T19:58:58.967191Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"When your notebook is run on the hidden test set, inference_server.serve must be called within 15 minutes of the notebook starting or the gateway will throw an error. If you need more than 15 minutes to load your model you can do so during the very first `predict` call, which does not have the usual 10 minute response deadline.","metadata":{"papermill":{"duration":0.002521,"end_time":"2024-10-10T11:58:33.6023","exception":false,"start_time":"2024-10-10T11:58:33.599779","status":"completed"},"tags":[]}},{"cell_type":"code","source":"inference_server = kaggle_evaluation.jane_street_inference_server.JSInferenceServer(predict)\n\nif os.getenv('KAGGLE_IS_COMPETITION_RERUN'):\n    inference_server.serve()\nelse:\n    inference_server.run_local_gateway(\n        (\n            '/kaggle/input/jane-street-realtime-marketdata-forecasting/test.parquet',\n            '/kaggle/input/jane-street-realtime-marketdata-forecasting/lags.parquet',\n        )\n    )","metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","papermill":{"duration":2.225871,"end_time":"2024-10-10T11:58:35.830964","exception":false,"start_time":"2024-10-10T11:58:33.605093","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2024-11-04T19:58:58.970202Z","iopub.execute_input":"2024-11-04T19:58:58.970627Z","iopub.status.idle":"2024-11-04T19:58:59.041114Z","shell.execute_reply.started":"2024-11-04T19:58:58.970581Z","shell.execute_reply":"2024-11-04T19:58:59.040209Z"},"trusted":true},"execution_count":null,"outputs":[]}]}