{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84493,"databundleVersionId":9871156,"sourceType":"competition"},{"sourceId":10059893,"sourceType":"datasetVersion","datasetId":6072331}],"dockerImageVersionId":30786,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport polars as pl\nimport kaggle_evaluation.jane_street_inference_server\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-11-30T15:39:56.455920Z","iopub.execute_input":"2024-11-30T15:39:56.456256Z","iopub.status.idle":"2024-11-30T15:39:58.517002Z","shell.execute_reply.started":"2024-11-30T15:39:56.456223Z","shell.execute_reply":"2024-11-30T15:39:58.515768Z"},"_kg_hide-input":true,"_kg_hide-output":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## **Define Feature and Target Columns**\nHere, we define the features used for training the model. `feature_columns` includes 79 features we will use for model prediction.","metadata":{}},{"cell_type":"code","source":"feature_columns = [f\"feature_{i:02}\" for i in range(79)]\ntarget_column = \"responder_6\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-30T15:39:58.519349Z","iopub.execute_input":"2024-11-30T15:39:58.519800Z","iopub.status.idle":"2024-11-30T15:39:58.525111Z","shell.execute_reply.started":"2024-11-30T15:39:58.519766Z","shell.execute_reply":"2024-11-30T15:39:58.524014Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## **Load the Trained Model**\nWe load the pre-trained XGBoost model saved as `xgboost_optimized_model.json`. This model is trained to predict the `responder_6` values based on the feature columns.","metadata":{}},{"cell_type":"code","source":"import xgboost as xgb\n\n# Load the trained model\nmodel = xgb.Booster()\nmodel.load_model(\"/kaggle/input/janestreet/xgboost_optimized_model.json\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-30T15:39:58.526507Z","iopub.execute_input":"2024-11-30T15:39:58.526842Z","iopub.status.idle":"2024-11-30T15:40:00.903624Z","shell.execute_reply.started":"2024-11-30T15:39:58.526810Z","shell.execute_reply":"2024-11-30T15:40:00.902563Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## **Define the Prediction Function**\nThe `predict` function is called by the inference server to make predictions on the test data. It uses the XGBoost model loaded above to generate predictions for the `responder_6` target.\n","metadata":{}},{"cell_type":"code","source":"def predict(test: pl.DataFrame, lags: pl.DataFrame | None) -> pl.DataFrame | pd.DataFrame:\n    \"\"\"Make a prediction using the XGBoost model.\"\"\"\n    \n    # ### Handling Lags Data\n    # The `lags` DataFrame provides data from the previous day at `time_id == 0`. We store it in a global variable\n    # `lags_` to use as extra features if needed in the future.\n    \n    global lags_\n    if lags is not None:\n        lags_ = lags\n\n    # ### Prepare Test Data for Prediction\n    # We select only the defined feature columns from the `test` DataFrame and convert them to a NumPy array.\n    # Any null values are filled with `-1` before passing the data to the model.\n    \n    X_test = test.select(feature_columns).fill_null(-1).to_numpy()\n\n    # ### Convert to XGBoost DMatrix\n    # To make predictions using XGBoost, the test data is converted into a DMatrix format and labeled with the \n    # appropriate feature names.\n    \n    dtest = xgb.DMatrix(X_test, feature_names=feature_columns)\n    \n    # ### Generate Predictions\n    # We use the XGBoost model to predict `responder_6` values based on the features in `dtest`.\n    \n    y_pred = model.predict(dtest)\n    \n    # ### Prepare Output DataFrame\n    # The output DataFrame is prepared with `row_id` from `test` and the predicted values in the `responder_6` column.\n    # This output DataFrame is returned to the inference server.\n    \n    predictions = test.select('row_id').with_columns(\n        pl.Series(\"responder_6\", y_pred)\n    )\n    \n    # ### Validate Output Format\n    # We assert the output format to ensure it meets the requirements, specifically that it is a DataFrame with\n    # columns `['row_id', 'responder_6']` and matches the row count of the input `test`.\n    \n    assert isinstance(predictions, (pl.DataFrame, pd.DataFrame)), \"Output must be a DataFrame\"\n    assert predictions.columns == ['row_id', 'responder_6'], \"Output columns must be ['row_id', 'responder_6']\"\n    assert len(predictions) == len(test), \"Output row count must match input row count\"\n\n    return predictions","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-30T15:40:00.916746Z","iopub.execute_input":"2024-11-30T15:40:00.917280Z","iopub.status.idle":"2024-11-30T15:40:00.925928Z","shell.execute_reply.started":"2024-11-30T15:40:00.917236Z","shell.execute_reply":"2024-11-30T15:40:00.924771Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## **Set Up the Inference Server**\nThis section initializes the inference server and determines whether to run in a competition rerun environment or a local environment. The server listens for test data batches and returns predictions as required.","metadata":{}},{"cell_type":"code","source":"inference_server = kaggle_evaluation.jane_street_inference_server.JSInferenceServer(predict)\n\n# ## Run the Inference Server\n# Depending on the environment (Kaggle competition rerun or local testing), the server runs accordingly.\n\nif os.getenv('KAGGLE_IS_COMPETITION_RERUN'):\n    inference_server.serve()\nelse:\n    inference_server.run_local_gateway(\n        (\n            '/kaggle/input/jane-street-real-time-market-data-forecasting/test.parquet',\n            '/kaggle/input/jane-street-real-time-market-data-forecasting/lags.parquet',\n        )\n    )","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-30T15:40:00.927297Z","iopub.execute_input":"2024-11-30T15:40:00.927609Z","iopub.status.idle":"2024-11-30T15:40:01.389374Z","shell.execute_reply.started":"2024-11-30T15:40:00.927578Z","shell.execute_reply":"2024-11-30T15:40:01.388442Z"}},"outputs":[],"execution_count":null}]}