{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"codemirror_mode":{"name":"ipython","version":3},"file_extension":".py","mimetype":"text/x-python","name":"python","nbconvert_exporter":"python","pygments_lexer":"ipython3","version":"3.10.14"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84493,"databundleVersionId":9871156,"sourceType":"competition"}],"dockerImageVersionId":30786,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false},"papermill":{"default_parameters":{},"duration":4.669361,"end_time":"2024-10-10T13:05:46.686069","environment_variables":{},"exception":null,"input_path":"__notebook__.ipynb","output_path":"__notebook__.ipynb","parameters":{},"start_time":"2024-10-10T13:05:42.016708","version":"2.6.0"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\n\nimport pandas as pd\nimport polars as pl\n\nimport kaggle_evaluation.jane_street_inference_server","metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","execution":{"iopub.status.busy":"2025-01-07T19:30:08.325997Z","iopub.execute_input":"2025-01-07T19:30:08.326398Z","iopub.status.idle":"2025-01-07T19:30:08.331699Z","shell.execute_reply.started":"2025-01-07T19:30:08.326364Z","shell.execute_reply":"2025-01-07T19:30:08.330478Z"},"papermill":{"duration":1.223703,"end_time":"2024-10-10T13:05:45.825911","exception":false,"start_time":"2024-10-10T13:05:44.602208","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data = pd.read_parquet('/kaggle/input/jane-street-real-time-market-data-forecasting/train.parquet/partition_id=0/part-0.parquet', engine='pyarrow')\ndata.describe()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-07T19:30:08.581769Z","iopub.execute_input":"2025-01-07T19:30:08.582135Z","iopub.status.idle":"2025-01-07T19:30:16.664478Z","shell.execute_reply.started":"2025-01-07T19:30:08.582104Z","shell.execute_reply":"2025-01-07T19:30:16.663259Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# symbol_ids = data['symbol_id'].unique()\n# dates = data['date_id'].unique()\n# counts = []\n\n# for symbol_id in symbol_ids:\n#     for date in dates:\n#         counts.append([symbol_id,date,data[(data['symbol_id'] == symbol_id) & (data['date_id'] == date)].shape[0]])\n\n# for values in counts:\n#     if values[2] != 849:\n#         print(f\"{values[0]} {values[1]}: {values[2]}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-07T19:30:16.666604Z","iopub.execute_input":"2025-01-07T19:30:16.667067Z","iopub.status.idle":"2025-01-07T19:30:16.672479Z","shell.execute_reply.started":"2025-01-07T19:30:16.667017Z","shell.execute_reply":"2025-01-07T19:30:16.671283Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# NOT ENOUGH MEMORY TO RUN THIS CELL\n# all_counts = data.apply(pd.Series.value_counts)\n# print(all_counts)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-07T19:30:16.674237Z","iopub.execute_input":"2025-01-07T19:30:16.674710Z","iopub.status.idle":"2025-01-07T19:30:16.685838Z","shell.execute_reply.started":"2025-01-07T19:30:16.674659Z","shell.execute_reply":"2025-01-07T19:30:16.684555Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"na_counts = data.isna().sum()\nna_counts","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-07T19:30:16.688610Z","iopub.execute_input":"2025-01-07T19:30:16.689112Z","iopub.status.idle":"2025-01-07T19:30:16.932384Z","shell.execute_reply.started":"2025-01-07T19:30:16.689062Z","shell.execute_reply":"2025-01-07T19:30:16.931210Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Removing columns that are null entirely\nnullColumns = []\n\nnullCols = data.columns[pd.isnull(data).all()].tolist()\nprint('Null Columns:',nullCols)\ndata = data.drop(nullCols, axis=1)\ndata.shape\n\nnanValues = data.isnull().sum()\ndata.dropna(inplace=True)\ndata.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-07T19:30:16.934061Z","iopub.execute_input":"2025-01-07T19:30:16.934473Z","iopub.status.idle":"2025-01-07T19:30:17.953007Z","shell.execute_reply.started":"2025-01-07T19:30:16.934427Z","shell.execute_reply":"2025-01-07T19:30:17.951894Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-07T19:30:17.955104Z","iopub.execute_input":"2025-01-07T19:30:17.955574Z","iopub.status.idle":"2025-01-07T19:30:18.177357Z","shell.execute_reply.started":"2025-01-07T19:30:17.955535Z","shell.execute_reply":"2025-01-07T19:30:18.176011Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import seaborn as sns\nimport matplotlib.pyplot as plt\n\n# Drop all responder columns except responder_6\nfeature_columns = data.columns.difference(['responder_0', 'responder_1', 'responder_2', 'responder_3', \n                                         'responder_4', 'responder_5', 'responder_7', 'responder_8', \n                                         'responder_6'])\n\n# Compute the correlation of features with responder_6\ncorr_with_responder6 = data[feature_columns].corrwith(data['responder_6'])\n\n# Visualize the correlation with a bar plot\nplt.figure(figsize=(10, 12))\nsns.barplot(y=corr_with_responder6.index, x=corr_with_responder6.values, palette='coolwarm')\nplt.title('Correlation of Features with responder_6')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-07T19:30:23.183880Z","iopub.execute_input":"2025-01-07T19:30:23.184228Z","iopub.status.idle":"2025-01-07T19:30:27.006334Z","shell.execute_reply.started":"2025-01-07T19:30:23.184199Z","shell.execute_reply":"2025-01-07T19:30:27.005144Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Set a threshold for low correlation (e.g., abs(correlation) < 0.05)\ncorr_threshold = 0.02\nlow_correlation_features = corr_with_responder6[abs(corr_with_responder6) < corr_threshold].index\n\n# Assuming low_correlation_features is a pandas.Index object\nlow_correlation_features = low_correlation_features.difference(['date_id', 'time_id', 'symbol_id'])\n\n\n# Drop low-correlation features from the DataFrame\ndata_cleaned = data.drop(columns=low_correlation_features)\n\nprint(f\"Features dropped due to low correlation with responder_6: {list(low_correlation_features)}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-07T19:30:48.378375Z","iopub.execute_input":"2025-01-07T19:30:48.378753Z","iopub.status.idle":"2025-01-07T19:30:48.436281Z","shell.execute_reply.started":"2025-01-07T19:30:48.378721Z","shell.execute_reply":"2025-01-07T19:30:48.435109Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Drop all responder columns except responder_6\n# data = df.drop(columns=['responder_0', 'responder_1', 'responder_2', 'responder_3', \n#                         'responder_4', 'responder_5', 'responder_7', 'responder_8'])\n\n# Set up the figure size for a large heatmap\nplt.figure(figsize=(20, 16))\n\n# Generate the heatmap for the dataset correlations\nsns.heatmap(data_cleaned.corr(), annot=False, cmap='coolwarm', linewidths=0.5)\n\n# Add title\nplt.title('Heatmap of Correlations for All Columns (Including responder_6)')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-07T19:30:34.767784Z","iopub.execute_input":"2025-01-07T19:30:34.768182Z","iopub.status.idle":"2025-01-07T19:30:36.400954Z","shell.execute_reply.started":"2025-01-07T19:30:34.768148Z","shell.execute_reply":"2025-01-07T19:30:36.399712Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"unique_responders=['responder_0', 'responder_1', 'responder_2', 'responder_3', \n                        'responder_4', 'responder_5', 'responder_6','responder_7', 'responder_8']\n\n# Get unique symbol_ids\nunique_symbols = data_cleaned['symbol_id'].unique()\n\n# Loop through each symbol_id and plot\nfor symbol in unique_symbols:\n    # Filter data for the current symbol_id\n    symbol_data = data_cleaned[(data_cleaned['symbol_id'] == symbol) & (data_cleaned['date_id'] == 0)]\n    \n    # Reset index for the symbol-specific data\n    symbol_data = symbol_data.reset_index(drop=True)\n    \n    # Plot for feature_05 and responder_6\n    plt.figure(figsize=(10, 6))\n    plt.plot(symbol_data.index, symbol_data['feature_06'], label='Feature 06')\n    plt.plot(symbol_data.index, symbol_data['responder_6'], label='Responder 6')\n    plt.xlabel('Index')\n    plt.ylabel('Value')\n    plt.title(f'Symbol ID: {symbol} - Feature 05 and Responder 6')\n    plt.legend()\n    plt.grid()\n    plt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-07T19:30:52.266116Z","iopub.execute_input":"2025-01-07T19:30:52.266559Z","iopub.status.idle":"2025-01-07T19:30:58.241271Z","shell.execute_reply.started":"2025-01-07T19:30:52.266520Z","shell.execute_reply":"2025-01-07T19:30:58.239990Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport matplotlib.pyplot as plt\n\n# Get unique symbols\nunique_symbols = data_cleaned['symbol_id'].unique()\n\n# List of responders\nresponders = [col for col in data_cleaned.columns if 'responder_' in col]\n\n# Initialize a dictionary to store correlations\ncorrelation_results = {}\n\nfor symbol in unique_symbols:\n    # Filter data for the current symbol\n    symbol_data = data_cleaned[data_cleaned['symbol_id'] == symbol]\n    \n    # Compute correlation of responders with responder_6\n    correlations = symbol_data[responders].corr()['responder_6'].drop('responder_6')\n    \n    # Store results\n    correlation_results[symbol] = correlations\n\n    # Plot correlations\n    plt.figure(figsize=(10, 6))\n    correlations.plot(kind='bar', color='skyblue', edgecolor='black')\n    plt.title(f'Correlation of Responders with Responder 6 (Symbol ID: {symbol})')\n    plt.xlabel('Responders')\n    plt.ylabel('Correlation')\n    plt.xticks(rotation=45)\n    plt.grid(axis='y')\n    plt.tight_layout()\n    plt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-05T19:04:19.938173Z","iopub.execute_input":"2025-01-05T19:04:19.938953Z","iopub.status.idle":"2025-01-05T19:04:26.716273Z","shell.execute_reply.started":"2025-01-05T19:04:19.938909Z","shell.execute_reply":"2025-01-05T19:04:26.715095Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"corr_responders=['responder_3', 'responder_7', 'responder_8', 'responder_6']\n\nselected_date_id = 1\n\n# Get unique symbol_ids\nunique_symbols = data_cleaned[(data_cleaned['date_id'] == selected_date_id)]['symbol_id'].unique()\n\n# Loop through each symbol_id and plot\nfor symbol in unique_symbols:\n    # Filter data for the current symbol_id\n    symbol_data = data_cleaned[(data_cleaned['symbol_id'] == symbol) & (data_cleaned['date_id'] == selected_date_id)]\n    \n    # Reset index for the symbol-specific data\n    symbol_data = symbol_data.reset_index(drop=True)\n    \n    # Plot for feature_05 and responder_6\n    plt.figure(figsize=(10, 6))\n    for responder in corr_responders:\n        plt.plot(symbol_data.index, symbol_data[responder], label=responder.replace('_',' ').capitalize())\n    plt.xlabel('Index')\n    plt.ylabel('Value')\n    plt.title(f'Symbol ID: {symbol} - Feature 05 and Responder 6')\n    plt.legend()\n    plt.grid()\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-05T19:09:24.081952Z","iopub.execute_input":"2025-01-05T19:09:24.082330Z","iopub.status.idle":"2025-01-05T19:09:27.855901Z","shell.execute_reply.started":"2025-01-05T19:09:24.082296Z","shell.execute_reply":"2025-01-05T19:09:27.854706Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"lags_data = pd.read_parquet('/kaggle/input/jane-street-real-time-market-data-forecasting/lags.parquet/date_id=0/part-0.parquet', engine='pyarrow')\nlags_data.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-27T20:05:35.974299Z","iopub.execute_input":"2024-12-27T20:05:35.974756Z","iopub.status.idle":"2024-12-27T20:05:36.007160Z","shell.execute_reply.started":"2024-12-27T20:05:35.974718Z","shell.execute_reply":"2024-12-27T20:05:36.005702Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"lags_data.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-27T20:05:19.862331Z","iopub.execute_input":"2024-12-27T20:05:19.863469Z","iopub.status.idle":"2024-12-27T20:05:19.886531Z","shell.execute_reply.started":"2024-12-27T20:05:19.863377Z","shell.execute_reply":"2024-12-27T20:05:19.885165Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"The evaluation API requires that you set up a server which will respond to inference requests. We have already defined the server; you just need write the predict function. When we evaluate your submission on the hidden test set the client defined in `jane_street_gateway` will run in a different container with direct access to the hidden test set and hand off the data timestep by timestep.\n\n\n\nYour code will always have access to the published copies of the files.","metadata":{"papermill":{"duration":0.002051,"end_time":"2024-10-10T13:05:45.83073","exception":false,"start_time":"2024-10-10T13:05:45.828679","status":"completed"},"tags":[]}},{"cell_type":"code","source":"features = pd.read_csv('/kaggle/input/jane-street-real-time-market-data-forecasting/features.csv')\nfeatures.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-27T18:43:23.554948Z","iopub.execute_input":"2024-12-27T18:43:23.555361Z","iopub.status.idle":"2024-12-27T18:43:23.583762Z","shell.execute_reply.started":"2024-12-27T18:43:23.555327Z","shell.execute_reply":"2024-12-27T18:43:23.582767Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"responders = pd.read_csv('/kaggle/input/jane-street-real-time-market-data-forecasting/responders.csv')\nresponders.head(10)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-27T18:45:28.399722Z","iopub.execute_input":"2024-12-27T18:45:28.401013Z","iopub.status.idle":"2024-12-27T18:45:28.417445Z","shell.execute_reply.started":"2024-12-27T18:45:28.400945Z","shell.execute_reply":"2024-12-27T18:45:28.416486Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from prophet import Prophet\nimport pandas as pd\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import r2_score\n\n# Assuming data_cleaned is already loaded and contains columns `date_id`, `time_id`, and `responder_6`\n# Step 1: Prepare the data\n# Use date_id as the base and assign sequential timestamps for the 'ds' column\ndata_cleaned['ds'] = pd.to_datetime(data_cleaned['date_id'], unit='D', origin='1970-01-01')\ndata_cleaned['y'] = data_cleaned['responder_6']\n# Assign weight column with value 1 for all rows\ndata_cleaned['weight'] = data['weight']\n\n# Step 2: Train-test split\ntrain, test = train_test_split(data_cleaned, test_size=0.2, shuffle=False)\n\n# Step 3: Initialize and configure the Prophet model\nmodel = Prophet(\n    yearly_seasonality=True,\n    weekly_seasonality=True,\n    daily_seasonality=False,\n    seasonality_mode='multiplicative'\n)\n\n# Step 4: Fit the model to the training data\nmodel.fit(train[['ds', 'y']])\n\n# Step 5: Make predictions on the test set\ntest['ds'] = pd.to_datetime(test['date_id'], unit='D', origin='1970-01-01')\npredictions = model.predict(test[['ds']])\ntest['yhat'] = predictions['yhat']\n\n# Step 6: Evaluate the model using R-squared\nnumerator = ((test['weight'] * (test['y'] - test['yhat']) ** 2).sum())\ndenominator = ((test['weight'] * (test['y'] ** 2)).sum())\nr2 = 1 - (numerator / denominator)\n\nprint(f\"R-squared: {r2}\")\n\n# Step 7: Define the predict function\ndef predict(test: pd.DataFrame) -> pd.DataFrame:\n    \"\"\"Make predictions using the Prophet model.\"\"\"\n    # Prepare the test data with sequential timestamps based on date_id\n    test['ds'] = pd.to_datetime(test['date_id'], unit='D', origin='1970-01-01')\n    # Use the model to predict\n    predictions = model.predict(test[['ds']])\n    # Return only the necessary columns: row_id and predicted responder_6 (yhat)\n    test['responder_6'] = predictions['yhat']\n    return test[['row_id', 'responder_6']]\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-07T20:04:53.643037Z","iopub.execute_input":"2025-01-07T20:04:53.643929Z","iopub.status.idle":"2025-01-07T20:16:10.989823Z","shell.execute_reply.started":"2025-01-07T20:04:53.643800Z","shell.execute_reply":"2025-01-07T20:16:10.988379Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Step 8: Make future predictions (e.g., for the next 6 months)\nfuture = model.make_future_dataframe(periods=1)  # 180 days (approximately 6 months)\nforecast = model.predict(future)\n\n# Step 9: Evaluate or visualize the forecast\nmodel.plot(forecast)\nmodel.plot_components(forecast)\n\n# Step 10: Save or return the forecast\nforecast[['ds', 'yhat', 'yhat_lower', 'yhat_upper']].to_csv('forecast_results.csv', index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-07T20:19:11.587947Z","iopub.execute_input":"2025-01-07T20:19:11.588370Z","iopub.status.idle":"2025-01-07T20:19:18.780468Z","shell.execute_reply.started":"2025-01-07T20:19:11.588335Z","shell.execute_reply":"2025-01-07T20:19:18.779138Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data_cleaned.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-07T20:28:25.007086Z","iopub.execute_input":"2025-01-07T20:28:25.007526Z","iopub.status.idle":"2025-01-07T20:28:25.110851Z","shell.execute_reply.started":"2025-01-07T20:28:25.007486Z","shell.execute_reply":"2025-01-07T20:28:25.109535Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import xgboost as xgb\nimport pandas as pd\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import r2_score\n\n# Assuming data_cleaned is already loaded and contains columns `date_id`, `time_id`, and `responder_6`\n# Step 1: Prepare the data\n# Assign weight column with value 1 for all rows\ndata_cleaned['weight'] = data['weight']\n\n# Use `date_id` and `time_id` as features along with other available features\nX = data_cleaned.drop(columns=['responder_6','ds','y'])  # Drop target and row identifier\ny = data_cleaned['responder_6']\n\n# Step 2: Train-test split\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, shuffle=False)\n\n# Step 3: Train the XGBoost model\nxgb_model = xgb.XGBRegressor(objective='reg:squarederror', n_estimators=100, learning_rate=0.1, max_depth=6, random_state=42)\nxgb_model.fit(X_train, y_train)\n\n# Step 4: Make predictions on the test set\ny_pred = xgb_model.predict(X_test)\n\n# Step 5: Evaluate the model using R-squared\nnumerator = ((X_test['weight'] * (y_test - y_pred) ** 2).sum())\ndenominator = ((X_test['weight'] * (y_test ** 2)).sum())\nr2 = 1 - (numerator / denominator)\n\nprint(f\"R-squared: {r2}\")\n\n# Step 6: Define the predict function\ndef predict(test: pd.DataFrame) -> pd.DataFrame:\n    \"\"\"Make predictions using the XGBoost model.\"\"\"\n    # Drop target columns or unnecessary columns from the test set\n    test_features = test.drop(columns=['responder_6'], errors='ignore')\n    # Use the model to predict\n    predictions = xgb_model.predict(test_features)\n    # Return only the necessary columns: row_id and predicted responder_6\n    test['responder_6'] = predictions\n    return test[['responder_6']]\n\n# Optional: Save or return predictions for further use\npredictions_df = predict(X_test)\npredictions_df.to_csv('predictions.csv', index=False)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-07T20:33:53.130379Z","iopub.execute_input":"2025-01-07T20:33:53.130852Z","iopub.status.idle":"2025-01-07T20:34:02.672668Z","shell.execute_reply.started":"2025-01-07T20:33:53.130814Z","shell.execute_reply":"2025-01-07T20:34:02.671082Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"lags_ : pl.DataFrame | None = None\n\n\n# Replace this function with your inference code.\n# You can return either a Pandas or Polars dataframe, though Polars is recommended.\n# Each batch of predictions (except the very first) must be returned within 1 minute of the batch features being provided.\ndef predict(test: pl.DataFrame, lags: pl.DataFrame | None) -> pl.DataFrame | pd.DataFrame:\n    \"\"\"Make a prediction.\"\"\"\n    # All the responders from the previous day are passed in at time_id == 0. We save them in a global variable for access at every time_id.\n    # Use them as extra features, if you like.\n    global lags_\n    if lags is not None:\n        lags_ = lags\n\n    # Replace this section with your own predictions\n    predictions = test.select(\n        'row_id',\n        pl.lit(0.0).alias('responder_6'),\n    )\n\n    if isinstance(predictions, pl.DataFrame):\n        assert predictions.columns == ['row_id', 'responder_6']\n    elif isinstance(predictions, pd.DataFrame):\n        assert (predictions.columns == ['row_id', 'responder_6']).all()\n    else:\n        raise TypeError('The predict function must return a DataFrame')\n    # Confirm has as many rows as the test data.\n    assert len(predictions) == len(test)\n\n    return predictions","metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","execution":{"iopub.status.busy":"2024-12-27T18:27:43.547855Z","iopub.execute_input":"2024-12-27T18:27:43.548340Z","iopub.status.idle":"2024-12-27T18:27:43.560419Z","shell.execute_reply.started":"2024-12-27T18:27:43.548294Z","shell.execute_reply":"2024-12-27T18:27:43.559201Z"},"papermill":{"duration":0.015917,"end_time":"2024-10-10T13:05:45.848958","exception":false,"start_time":"2024-10-10T13:05:45.833041","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"When your notebook is run on the hidden test set, inference_server.serve must be called within 15 minutes of the notebook starting or the gateway will throw an error. If you need more than 15 minutes to load your model you can do so during the very first `predict` call, which does not have the usual 1 minute response deadline.","metadata":{"papermill":{"duration":0.00196,"end_time":"2024-10-10T13:05:45.853279","exception":false,"start_time":"2024-10-10T13:05:45.851319","status":"completed"},"tags":[]}},{"cell_type":"code","source":"inference_server = kaggle_evaluation.jane_street_inference_server.JSInferenceServer(predict)\n\nif os.getenv('KAGGLE_IS_COMPETITION_RERUN'):\n    inference_server.serve()\nelse:\n    inference_server.run_local_gateway(\n        (\n            '/kaggle/input/jane-street-real-time-market-data-forecasting/test.parquet',\n            '/kaggle/input/jane-street-real-time-market-data-forecasting/lags.parquet',\n        )\n    )","metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","execution":{"iopub.status.busy":"2024-12-27T18:27:48.768001Z","iopub.execute_input":"2024-12-27T18:27:48.768888Z","iopub.status.idle":"2024-12-27T18:27:49.073575Z","shell.execute_reply.started":"2024-12-27T18:27:48.768849Z","shell.execute_reply":"2024-12-27T18:27:49.072642Z"},"papermill":{"duration":0.308219,"end_time":"2024-10-10T13:05:46.163573","exception":false,"start_time":"2024-10-10T13:05:45.855354","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}