{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84493,"databundleVersionId":9871156,"sourceType":"competition"},{"sourceId":206429546,"sourceType":"kernelVersion"},{"sourceId":209080553,"sourceType":"kernelVersion"}],"dockerImageVersionId":30786,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Submission\n\nThis is a notebook intended for inference. [Also check out the training notebook](https://www.kaggle.com/code/regisvargas/jane-street-a-beginner-s-notebook).\n\nFor details about the submission process, see [Jane Street RMF Demo Submission](https://www.kaggle.com/code/ryanholbrook/jane-street-rmf-demo-submission) notebook.","metadata":{}},{"cell_type":"code","source":"Is_keras = True","metadata":{"execution":{"iopub.status.busy":"2024-11-23T09:13:34.177638Z","iopub.execute_input":"2024-11-23T09:13:34.181126Z","iopub.status.idle":"2024-11-23T09:13:34.226132Z","shell.execute_reply.started":"2024-11-23T09:13:34.181054Z","shell.execute_reply":"2024-11-23T09:13:34.223506Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from tensorflow.keras.models import load_model\n# Load the saved model\nif Is_keras:\n    model = load_model(\"/kaggle/input/jane-street-a-beginner-s-notebook/model.keras\")","metadata":{"execution":{"iopub.status.busy":"2024-11-23T09:13:34.227924Z","iopub.execute_input":"2024-11-23T09:13:34.228340Z","iopub.status.idle":"2024-11-23T09:13:49.065041Z","shell.execute_reply.started":"2024-11-23T09:13:34.228293Z","shell.execute_reply":"2024-11-23T09:13:49.064096Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"encoder = load_model(\"/kaggle/input/jane-street-a-beginner-s-notebook/pretrained_encoder.keras\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T09:13:49.066247Z","iopub.execute_input":"2024-11-23T09:13:49.066819Z","iopub.status.idle":"2024-11-23T09:13:49.130208Z","shell.execute_reply.started":"2024-11-23T09:13:49.066785Z","shell.execute_reply":"2024-11-23T09:13:49.128808Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import joblib\nmodel_xgb = joblib.load(\"/kaggle/input/jane-street-a-beginner-s-notebook/xgboost_sklearn.pkl\")","metadata":{"execution":{"iopub.status.busy":"2024-11-23T09:13:49.132682Z","iopub.execute_input":"2024-11-23T09:13:49.133075Z","iopub.status.idle":"2024-11-23T09:13:50.449808Z","shell.execute_reply.started":"2024-11-23T09:13:49.133042Z","shell.execute_reply":"2024-11-23T09:13:50.448621Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport polars as pl\nimport kaggle_evaluation.jane_street_inference_server","metadata":{"execution":{"iopub.status.busy":"2024-11-23T09:13:50.451872Z","iopub.execute_input":"2024-11-23T09:13:50.452547Z","iopub.status.idle":"2024-11-23T09:13:50.945926Z","shell.execute_reply.started":"2024-11-23T09:13:50.452511Z","shell.execute_reply":"2024-11-23T09:13:50.944591Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import polars as pl\nimport numpy as np\nimport pandas as pd\n\nlets_see = pd.read_parquet('/kaggle/input/jane-street-a-beginner-s-notebook/submission.parquet')\nlets_see","metadata":{"execution":{"iopub.status.busy":"2024-11-23T09:13:50.947390Z","iopub.execute_input":"2024-11-23T09:13:50.947785Z","iopub.status.idle":"2024-11-23T09:13:51.345314Z","shell.execute_reply.started":"2024-11-23T09:13:50.947738Z","shell.execute_reply":"2024-11-23T09:13:51.344250Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"lets_see_2 = pd.read_parquet('/kaggle/input/v10-of-jane-street-a-beginner-s-notebook/submission.parquet')\nlets_see_2","metadata":{"execution":{"iopub.status.busy":"2024-11-23T09:13:51.346410Z","iopub.execute_input":"2024-11-23T09:13:51.346708Z","iopub.status.idle":"2024-11-23T09:13:51.363773Z","shell.execute_reply.started":"2024-11-23T09:13:51.346680Z","shell.execute_reply":"2024-11-23T09:13:51.362780Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Assuming `model` is your trained model\n# Assuming features required by the model are named 'feature_00', 'feature_01', etc.\ndef predict(test: pl.DataFrame, lags: pl.DataFrame | None) -> pl.DataFrame | pd.DataFrame:\n    \"\"\"Make a prediction.\"\"\"\n    global lags_\n    if lags is not None:\n        lags_ = lags\n    # Extract the features for the model input\n    feature_columns = [col for col in test.columns if col.startswith(\"feature_\")]\n    features = test.select(feature_columns).to_numpy()  # Convert to numpy array for model input\n    features = np.nan_to_num(features, nan=0.0, posinf=0.0, neginf=0.0)\n    # Generate predictions using the model\n    #model_predictions = model.predict(features)\n    if Is_keras:    \n        responder_6_predictions = model.predict(features)[:,0]\n    else:\n        responder_6_predictions = model_xgb.predict(features)\n    #responder_6_predictions = model_predictions[:, 6]  # Assuming responder_6 is at index 6\n    # Create a new Polars DataFrame with row_id and responder_6 predictions\n    predictions = test.select(\"row_id\").with_columns(\n        pl.Series(\"responder_6\", responder_6_predictions)\n    )\n    # Ensure the output format and length requirements\n    if isinstance(predictions, pl.DataFrame):\n        assert predictions.columns == ['row_id', 'responder_6']\n    elif isinstance(predictions, pd.DataFrame):\n        assert (predictions.columns == ['row_id', 'responder_6']).all()\n    else:\n        raise TypeError('The predict function must return a DataFrame')\n    \n    assert len(predictions) == len(test)\n    print(predictions)\n    return predictions","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T09:13:51.365117Z","iopub.execute_input":"2024-11-23T09:13:51.365516Z","iopub.status.idle":"2024-11-23T09:13:51.374650Z","shell.execute_reply.started":"2024-11-23T09:13:51.365483Z","shell.execute_reply":"2024-11-23T09:13:51.373603Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"inference_server = kaggle_evaluation.jane_street_inference_server.JSInferenceServer(predict)\nif os.getenv('KAGGLE_IS_COMPETITION_RERUN'):\n    inference_server.serve()\nelse:\n    inference_server.run_local_gateway(\n        (\n            '/kaggle/input/jane-street-real-time-market-data-forecasting/test.parquet',\n            '/kaggle/input/jane-street-real-time-market-data-forecasting/lags.parquet',\n        )\n    )","metadata":{"execution":{"iopub.status.busy":"2024-11-23T09:13:51.376488Z","iopub.execute_input":"2024-11-23T09:13:51.377220Z","iopub.status.idle":"2024-11-23T09:13:52.058561Z","shell.execute_reply.started":"2024-11-23T09:13:51.377173Z","shell.execute_reply":"2024-11-23T09:13:52.057170Z"},"trusted":true},"outputs":[],"execution_count":null}]}