{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84493,"databundleVersionId":9871156,"sourceType":"competition"},{"sourceId":9900119,"sourceType":"datasetVersion","datasetId":6072331}],"dockerImageVersionId":30786,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport polars as pl\nimport kaggle_evaluation.jane_street_inference_server\nimport tensorflow as tf\nimport pickle\n\nimport warnings\nwarnings.filterwarnings(\"ignore\")\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-11-14T16:05:41.765699Z","iopub.execute_input":"2024-11-14T16:05:41.766175Z","iopub.status.idle":"2024-11-14T16:05:41.804579Z","shell.execute_reply.started":"2024-11-14T16:05:41.766064Z","shell.execute_reply":"2024-11-14T16:05:41.803064Z"},"_kg_hide-output":true,"_kg_hide-input":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## **Goal of the Competition**\n\nIn this competition, hosted by Jane Street, you'll build a model using real-world data derived from production systems, which offers a glimpse into the daily challenges of successful trading. This challenge highlights the difficulties in modeling financial markets, including fat-tailed distributions, non-stationary time series, and sudden shifts in market behavior.\n\n**I will continue to work on and update this notebook. Please upvote it if you find it useful in this interesting challenge!**\n\n## **Data Loading and Preparation**\n\nWe will start by loading the `train.parquet` file and preparing it for analysis.\n\n**train.parquet**  \n  The training set, contains historical data and returns. For convenience, the training set has been partitioned into ten parts.\n\n- **date_id** and **time_id**  \n  Integer values that are ordinally sorted, providing a chronological structure to the data, although the actual time intervals between `time_id` values may vary.\n\n- **symbol_id**  \n  Identifies a unique financial instrument.\n\n- **weight**  \n  The weighting used for calculating the scoring function.\n\n- **feature_{00...78}**  \n  Anonymized market data.\n\n- **responder_{0...8}**  \n  Anonymized responders clipped between -5 and 5. The **responder_6** field is what you are trying to predict.\n","metadata":{}},{"cell_type":"code","source":"# Set pandas option to display all columns\npd.set_option('display.max_columns', None)\n\n# Load the Parquet file\nfile_path = '/kaggle/input/jane-street-real-time-market-data-forecasting/train.parquet/partition_id=4/part-0.parquet'\ndata = pd.read_parquet(file_path)\ndata","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-14T16:05:41.815290Z","iopub.execute_input":"2024-11-14T16:05:41.815775Z","iopub.status.idle":"2024-11-14T16:05:44.681285Z","shell.execute_reply.started":"2024-11-14T16:05:41.815722Z","shell.execute_reply":"2024-11-14T16:05:44.680000Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## **Exploratory Data Analysis**","metadata":{}},{"cell_type":"code","source":"# Check for missing values\nprint(\"\\nMissing values per column:\")\nmissing_values = data.isnull().sum()\ndisplay(missing_values[missing_values > 0])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-14T16:05:44.690105Z","iopub.execute_input":"2024-11-14T16:05:44.690471Z","iopub.status.idle":"2024-11-14T16:05:45.315907Z","shell.execute_reply.started":"2024-11-14T16:05:44.690432Z","shell.execute_reply":"2024-11-14T16:05:45.314474Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns\n\n# Distribution of `responder_6` (target variable)\nplt.figure(figsize=(10, 6))\nsns.histplot(data['responder_6'].dropna(), kde=True, bins=30)\nplt.title(\"Distribution of responder_6 (Target Variable)\")\nplt.xlabel(\"Responder 6\")\nplt.ylabel(\"Frequency\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-14T16:05:45.327654Z","iopub.execute_input":"2024-11-14T16:05:45.328492Z","iopub.status.idle":"2024-11-14T16:06:11.203504Z","shell.execute_reply.started":"2024-11-14T16:05:45.328427Z","shell.execute_reply":"2024-11-14T16:06:11.202220Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Checking distributions for a few features\nfeature_cols = [f'feature_{i:02}' for i in range(5)]  # select first 5 features for visualization\nfig, axs = plt.subplots(1, 5, figsize=(20, 4))\nfor i, col in enumerate(feature_cols):\n    sns.histplot(data[col].dropna(), ax=axs[i], kde=True, bins=30)\n    axs[i].set_title(f\"Distribution of {col}\")\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-14T16:06:11.582675Z","iopub.execute_input":"2024-11-14T16:06:11.583709Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Time series plot for `responder_6` over `date_id` to check for temporal trends\nplt.figure(figsize=(15, 6))\nsns.lineplot(x='date_id', y='responder_6', data=data)\nplt.title(\"Responder 6 over Time (date_id)\")\nplt.xlabel(\"Date ID\")\nplt.ylabel(\"Responder 6\")\nplt.show()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Select only the feature columns and responder_6\nfeature_columns = [f'feature_{i:02}' for i in range(79)]\nselected_data = data[feature_columns + ['responder_6']]\n\n# Calculate correlation between each feature and responder_6 individually\ncorrelations = selected_data[feature_columns].apply(lambda x: x.corr(data['responder_6'])).sort_values(ascending=False)\n\n# Display correlations\nprint(\"Correlation between features and responder_6:\")\ndisplay(correlations)\n\n# Plot the correlation values with rotated axes\nplt.figure(figsize=(10, 12))\ncorrelations.plot(kind='barh', color='skyblue')\nplt.title(\"Correlation of Features with Responder 6\")\nplt.xlabel(\"Correlation\")\nplt.ylabel(\"Features\")\nplt.gca().invert_yaxis()  \nplt.show()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## **Feature Engineering**\n\nTo enhance the predictive power of the model, we’ve applied various feature engineering techniques to generate new features from the original dataset. Below is a summary of each created feature and its purpose:\n\n1. **Moving Averages**  \n   - *Feature Format*: `feature_{i}_ma{window_size}`\n   - *Description*: Calculates a moving average over a specified window (e.g., 5 days) for each feature, smoothing out short-term fluctuations to capture trends. This helps in identifying underlying patterns over time.\n\n2. **Lag Features**  \n   - *Feature Format*: `feature_{i}_lag{lag_days}`\n   - *Description*: Adds a lagged version of each feature, representing values from previous observations (e.g., 1 day prior). Lag features capture past behavior, helping models understand historical context and dependencies.\n\n3. **Volatility (Rolling Standard Deviation)**  \n   - *Feature Format*: `feature_{i}_volatility`\n   - *Description*: Calculates the rolling standard deviation, measuring the variability of each feature within the specified window. This provides insights into periods of high activity or market volatility.\n\n4. **Relative Changes (Percent Change)**  \n   - *Feature Format*: `feature_{i}_pct_change`\n   - *Description*: Computes the percent change between consecutive observations for each feature, capturing the momentum or rate of change. This helps detect accelerating or decelerating trends.\n\n5. **Rolling Min and Max**  \n   - *Feature Format*: `feature_{i}_rolling_min`, `feature_{i}_rolling_max`\n   - *Description*: Computes the rolling minimum and maximum for each feature over a specified window, capturing the range of values. This helps identify extreme values and a feature's overall range over time.\n\nEach engineered feature is designed to enrich the dataset with additional contextual information, potentially improving the model’s ability to predict `responder_6`.\n","metadata":{}},{"cell_type":"code","source":"from tqdm import tqdm\n\n# Set this switch to True to enable feature engineering, False to disable it\nrun_feature_engineering = False\n\nif run_feature_engineering:\n    # 1. Moving Averages: Calculate moving averages for selected features to capture trends\n    window_size = 5  # Adjust the window size as needed\n    for i in tqdm(range(79), desc=\"Calculating Moving Averages\"):\n        feature_name = f'feature_{i:02}'\n        data[f'{feature_name}_ma{window_size}'] = data.groupby('symbol_id')[feature_name].transform(lambda x: x.rolling(window_size).mean())\n\n    # 2. Lag Features: Create lagged versions of each feature to capture past behavior\n    lag_days = 1  # Adjust the lag days as needed\n    for i in tqdm(range(79), desc=\"Creating Lag Features\"):\n        feature_name = f'feature_{i:02}'\n        data[f'{feature_name}_lag{lag_days}'] = data.groupby('symbol_id')[feature_name].shift(lag_days)\n\n    # 3. Volatility: Calculate rolling standard deviation to capture volatility for each feature\n    for i in tqdm(range(79), desc=\"Calculating Volatility\"):\n        feature_name = f'feature_{i:02}'\n        data[f'{feature_name}_volatility'] = data.groupby('symbol_id')[feature_name].transform(lambda x: x.rolling(window_size).std())\n\n    # 4. Relative Changes: Calculate percent changes for features to capture momentum\n    for i in tqdm(range(79), desc=\"Calculating Relative Changes\"):\n        feature_name = f'feature_{i:02}'\n        data[f'{feature_name}_pct_change'] = data.groupby('symbol_id')[feature_name].pct_change()\n\n    # 5. Rolling Min and Max: Calculate rolling minimum and maximum values to capture ranges over a window\n    for i in tqdm(range(79), desc=\"Calculating Rolling Min and Max\"):\n        feature_name = f'feature_{i:02}'\n        data[f'{feature_name}_rolling_min'] = data.groupby('symbol_id')[feature_name].transform(lambda x: x.rolling(window_size).min())\n        data[f'{feature_name}_rolling_max'] = data.groupby('symbol_id')[feature_name].transform(lambda x: x.rolling(window_size).max())","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"del data","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Plotting Moving Averages for a selected feature\n#feature = 'feature_00'\n#window_size = 5\n#plt.figure(figsize=(12, 6))\n#sns.lineplot(data=data, x='date_id', y=f'{feature}_ma{window_size}', hue='symbol_id', legend=None)\n#plt.title(f'Moving Average (Window={window_size}) of {feature}')\n#plt.xlabel('Date ID')\n#plt.ylabel(f'{feature} Moving Average')\n#plt.show()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Plotting Lagged Features for a selected feature\n#lag_days = 1\n#plt.figure(figsize=(12, 6))\n#sns.lineplot(data=data, x='date_id', y=f'{feature}_lag{lag_days}', hue='symbol_id', legend=None)\n#plt.title(f'Lagged Feature (Lag={lag_days} Day) of {feature}')\n#plt.xlabel('Date ID')\n#plt.ylabel(f'{feature} Lagged by {lag_days} Day')\n#plt.show()","metadata":{"trusted":true,"_kg_hide-input":true,"_kg_hide-output":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Plotting Volatility (Rolling Std Dev) for a selected feature\n#plt.figure(figsize=(12, 6))\n#sns.lineplot(data=data, x='date_id', y=f'{feature}_volatility', hue='symbol_id', legend=None)\n#plt.title(f'Volatility (Rolling Std Dev) of {feature}')\n#plt.xlabel('Date ID')\n#plt.ylabel(f'{feature} Volatility')\n#plt.show()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Plotting Relative Change for a selected feature\n#plt.figure(figsize=(12, 6))\n#sns.lineplot(data=data, x='date_id', y=f'{feature}_pct_change', hue='symbol_id', legend=None)\n#plt.title(f'Percent Change of {feature}')\n#plt.xlabel('Date ID')\n#plt.ylabel(f'{feature} Percent Change')\n#plt.show()","metadata":{"trusted":true,"_kg_hide-input":true,"_kg_hide-output":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## **Machine Learning and Submitting Predictions**\n\nWe will now load a trained XGBoost model to predict responder_6 without feature engineering. We will only use the default features provided.","metadata":{}},{"cell_type":"code","source":"feature_columns = [f\"feature_{i:02}\" for i in range(79)]\ntarget_column = \"responder_6\"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import xgboost as xgb\n\n# Load the trained model\nmodel = xgb.Booster()\nmodel.load_model(\"/kaggle/input/janestreet/xgboost_optimized_model.json\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Define the prediction function\ndef predict(test: pl.DataFrame, lags: pl.DataFrame | None) -> pl.DataFrame | pd.DataFrame:\n    \"\"\"Make a prediction using the XGBoost model.\"\"\"\n    global lags_\n    \n    # Update the global lags if provided\n    if lags is not None:\n        lags_ = lags\n\n    # Select feature columns and convert to a numpy array, filling null values with -1\n    X_test = test.select(feature_columns).fill_null(-1).to_numpy()\n\n    # Convert the test data to DMatrix format for XGBoost prediction, specifying feature names\n    dtest = xgb.DMatrix(X_test, feature_names=feature_columns)\n    \n    # Make predictions using the XGBoost model\n    y_pred = model.predict(dtest)\n    \n    # Prepare the predictions DataFrame\n    predictions = test.select('row_id').with_columns(\n        pl.Series(\"responder_6\", y_pred)\n    )\n    \n    # Ensure output format compliance\n    assert isinstance(predictions, (pl.DataFrame, pd.DataFrame)), \"Output must be a DataFrame\"\n    assert predictions.columns == ['row_id', 'responder_6'], \"Output columns must be ['row_id', 'responder_6']\"\n    assert len(predictions) == len(test), \"Output row count must match input row count\"\n\n    return predictions\n\n# Set up the inference server\ninference_server = kaggle_evaluation.jane_street_inference_server.JSInferenceServer(predict)\n\n# Run the server\nif os.getenv('KAGGLE_IS_COMPETITION_RERUN'):\n    inference_server.serve()\nelse:\n    inference_server.run_local_gateway(\n        (\n            '/kaggle/input/jane-street-real-time-market-data-forecasting/test.parquet',\n            '/kaggle/input/jane-street-real-time-market-data-forecasting/lags.parquet',\n        )\n    )","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}