{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84493,"databundleVersionId":9871156,"sourceType":"competition"}],"dockerImageVersionId":30822,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-19T09:40:06.647545Z","iopub.execute_input":"2024-12-19T09:40:06.648044Z","iopub.status.idle":"2024-12-19T09:40:06.709218Z","shell.execute_reply.started":"2024-12-19T09:40:06.648002Z","shell.execute_reply":"2024-12-19T09:40:06.708163Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport joblib\nimport polars as pl\nimport xgboost as xgb\nimport numpy as np\nimport pandas as pd\nimport kaggle_evaluation.jane_street_inference_server\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T09:55:05.537224Z","iopub.execute_input":"2024-12-19T09:55:05.538121Z","iopub.status.idle":"2024-12-19T09:55:05.544031Z","shell.execute_reply.started":"2024-12-19T09:55:05.538037Z","shell.execute_reply":"2024-12-19T09:55:05.542584Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Paths and constants\ninput_path = '/kaggle/input/jane-street-real-time-market-data-forecasting'\ndef read_selected_data(input_path):\n    # Define the directory containing your data files\n\n    # List three specific Parquet files you want to read\n    selected_files = [f\"partition_id={i}/part-0.parquet\" for i in range(1)]\n    # Load and filter the data from only the selected Parquet files\n    dfs = []\n    for file_name in selected_files:\n        file_path = f'{input_path}/train.parquet/{file_name}'\n        lazy_df = pl.scan_parquet(file_path)\n        df = lazy_df.collect()\n        dfs.append(df)\n\n    # Concatenate all dataframes into a single dataframe\n    full_df = pl.concat(dfs)\n\n    return full_df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T09:55:06.668029Z","iopub.execute_input":"2024-12-19T09:55:06.668365Z","iopub.status.idle":"2024-12-19T09:55:06.674087Z","shell.execute_reply.started":"2024-12-19T09:55:06.668339Z","shell.execute_reply":"2024-12-19T09:55:06.672874Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df = read_selected_data(input_path)\ndf = df.fill_null(strategy='forward')\n\n# Prepare feature names\nfeature_names = [f\"feature_{i:02d}\" for i in range(79)]\n\n# Prepare training and validation data\nnum_valid_dates = 180\ndates = df['date_id'].unique().to_numpy()\ntrain_dates = dates[-num_valid_dates:]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T09:55:15.868957Z","iopub.execute_input":"2024-12-19T09:55:15.869308Z","iopub.status.idle":"2024-12-19T09:55:16.735374Z","shell.execute_reply.started":"2024-12-19T09:55:15.869280Z","shell.execute_reply":"2024-12-19T09:55:16.734142Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_pd = df.to_pandas()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T09:55:18.632146Z","iopub.execute_input":"2024-12-19T09:55:18.632476Z","iopub.status.idle":"2024-12-19T09:55:19.626092Z","shell.execute_reply.started":"2024-12-19T09:55:18.632450Z","shell.execute_reply":"2024-12-19T09:55:19.624670Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_pd_last180 = df_pd.tail(180)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T09:55:21.119747Z","iopub.execute_input":"2024-12-19T09:55:21.120156Z","iopub.status.idle":"2024-12-19T09:55:21.125256Z","shell.execute_reply.started":"2024-12-19T09:55:21.120124Z","shell.execute_reply":"2024-12-19T09:55:21.124046Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pre_mid = df_pd_last180['responder_6'].median()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T09:55:23.020579Z","iopub.execute_input":"2024-12-19T09:55:23.021030Z","iopub.status.idle":"2024-12-19T09:55:23.027720Z","shell.execute_reply.started":"2024-12-19T09:55:23.020998Z","shell.execute_reply":"2024-12-19T09:55:23.026350Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pre_mid","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T09:55:24.746258Z","iopub.execute_input":"2024-12-19T09:55:24.746617Z","iopub.status.idle":"2024-12-19T09:55:24.753431Z","shell.execute_reply.started":"2024-12-19T09:55:24.746588Z","shell.execute_reply":"2024-12-19T09:55:24.752095Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Global lags storage\nlags_: pl.DataFrame | None = None\ndef predict(test: pl.DataFrame, lags: pl.DataFrame | None) -> pl.DataFrame:\n    global lags_\n    \n    # Logic for saving or loading lags\n    if lags is not None:\n        lags_ = lags\n    \n    test = test.to_pandas()\n    test['pre'] = pre_mid\n\n    output_df = pd.DataFrame({\"row_id\": test['row_id'], \"responder_6\": test['pre']})\n\n        \n    return pl.from_pandas(output_df)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T09:55:26.282277Z","iopub.execute_input":"2024-12-19T09:55:26.282621Z","iopub.status.idle":"2024-12-19T09:55:26.288580Z","shell.execute_reply.started":"2024-12-19T09:55:26.282591Z","shell.execute_reply":"2024-12-19T09:55:26.287348Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Setup the inference server\ninference_server = kaggle_evaluation.jane_street_inference_server.JSInferenceServer(predict)\n\n# Running the inference server\nif os.getenv('KAGGLE_IS_COMPETITION_RERUN'):\n    inference_server.serve()\nelse:\n    inference_server.run_local_gateway((\n        '/kaggle/input/jane-street-real-time-market-data-forecasting/test.parquet',\n        '/kaggle/input/jane-street-real-time-market-data-forecasting/lags.parquet',\n    ))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T09:55:28.293117Z","iopub.execute_input":"2024-12-19T09:55:28.293459Z","iopub.status.idle":"2024-12-19T09:55:28.496394Z","shell.execute_reply.started":"2024-12-19T09:55:28.293430Z","shell.execute_reply":"2024-12-19T09:55:28.495136Z"}},"outputs":[],"execution_count":null}]}