{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":84493,"databundleVersionId":9871156,"sourceType":"competition"}],"dockerImageVersionId":30786,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# 🧠 Real-Time Market Data Forecasting with LightGBM  \nWelcome to this **Kaggle Notebook**, where we tackle the *Jane Street Real-Time Market Data Forecasting* competition. Our goal is to predict `responder_6` using complex and anonymized financial datasets, leveraging state-of-the-art tools and machine learning techniques.\n\n## Highlights of This Notebook  \n- **Efficient Data Processing**: Polars is used to handle large parquet files swiftly.  \n- **Advanced Modeling**: LightGBM (GPU-powered) ensures scalable and robust regression modeling.  \n- **Incremental Training**: A custom `train_in_chunks` function manages training on large datasets.  \n- **Real-Time Predictions**: An end-to-end inference pipeline is set up for seamless prediction generation.  \n\n### Workflow  \n1. **Data Preparation**: Load and preprocess massive financial datasets.  \n2. **Feature Engineering**: Select and clean features for meaningful insights.  \n3. **Model Training**: Train LightGBM incrementally for better memory usage.  \n4. **Evaluation**: Calculate metrics like RMSE and accuracy for performance tracking.  \n5. **Submission Pipeline**: Generate predictions for competition submission.  \n\n### Tools and Libraries  \n- **Polars & Pandas**: For efficient data manipulation.  \n- **LightGBM**: Gradient boosting for regression tasks.  \n- **Scikit-learn**: Model evaluation and preprocessing.  \n- **TQDM**: Real-time progress visualization during training.\n\nStay tuned as we navigate through the complexities of financial forecasting while striving for accuracy and scalability! 🚀","metadata":{"_uuid":"8aee3e92-a0b4-4d9e-a374-76d94b90575c","_cell_guid":"5d94cb71-8751-428d-bca2-3f6c40fe2ed7","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false}}},{"cell_type":"markdown","source":"# Imports","metadata":{"_uuid":"11ae29fe-7d39-4da2-af23-6dc25717bdb0","_cell_guid":"69a06f02-1b13-456b-b274-62fd04dd60ab","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false}}},{"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\nimport polars as pl\nfrom tqdm import tqdm\nimport lightgbm as lgb\nimport matplotlib.pyplot as plt\nfrom lightgbm import LGBMRegressor\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.model_selection import train_test_split\nimport kaggle_evaluation.jane_street_inference_server\nfrom sklearn.metrics import r2_score , mean_squared_error","metadata":{"_uuid":"56d06f04-5057-494a-8074-d43b9ce9e4da","_cell_guid":"6b912fd8-e0c5-4834-8594-15d314c0f958","trusted":true,"collapsed":false,"execution":{"iopub.status.busy":"2024-11-18T17:11:09.664969Z","iopub.execute_input":"2024-11-18T17:11:09.665268Z","iopub.status.idle":"2024-11-18T17:11:16.827278Z","shell.execute_reply.started":"2024-11-18T17:11:09.665239Z","shell.execute_reply":"2024-11-18T17:11:16.826603Z"},"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import logging\nlogging.getLogger('lightgbm').setLevel(logging.WARNING)","metadata":{"_uuid":"088effad-65b6-48ac-b149-fc78d51a855b","_cell_guid":"483e03e1-1057-4110-96a6-2397edda53e3","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Set up paths for data files","metadata":{"_uuid":"a08ef5aa-7840-41a0-8de6-6039ce588867","_cell_guid":"44f0cc01-a9bf-469c-80d4-a412e33d8cd3","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false}}},{"cell_type":"code","source":"data_dir = \"/kaggle/input/jane-street-real-time-market-data-forecasting\"\ntrain_file = data_dir + \"/train.parquet\"\ntest_file = \"/kaggle/input/jane-street-real-time-market-data-forecasting/test.parquet/date_id=0/part-0.parquet\"\nlags_file = data_dir + \"/lags.parquet\"\nfeatures_file = data_dir + \"/features.csv\"\nresponders_file = data_dir + \"/responders.csv\"\nsample_submission_file = data_dir + \"/sample_submission.csv\"","metadata":{"_uuid":"3e234d3a-b678-4bba-8d75-16b00b4b6f7a","_cell_guid":"fa968832-9d3a-4369-932a-5ce5a339058c","trusted":true,"collapsed":false,"execution":{"iopub.status.busy":"2024-11-18T17:11:16.829031Z","iopub.execute_input":"2024-11-18T17:11:16.829650Z","iopub.status.idle":"2024-11-18T17:11:16.834158Z","shell.execute_reply.started":"2024-11-18T17:11:16.829609Z","shell.execute_reply":"2024-11-18T17:11:16.833314Z"},"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df = pl.scan_parquet(train_file)\n# Concatenate all partitions into a single DataFrame\ndf = df.collect()\nprint(df)","metadata":{"_uuid":"d3019c6e-b5d9-46ad-b284-aabac551bfeb","_cell_guid":"09b8681e-d8a3-410c-89dc-90e2af7724d7","trusted":true,"collapsed":false,"execution":{"iopub.status.busy":"2024-11-18T17:11:16.835092Z","iopub.execute_input":"2024-11-18T17:11:16.835309Z","iopub.status.idle":"2024-11-18T17:12:10.650062Z","shell.execute_reply.started":"2024-11-18T17:11:16.835287Z","shell.execute_reply":"2024-11-18T17:12:10.649109Z"},"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# column_names = df.columns\n\n# # Print the column names\n# print(column_names)","metadata":{"_uuid":"6a084ec0-8dad-4b44-8dde-866299ec8d02","_cell_guid":"8346a3ea-0126-45ca-952f-1672d7ba02c1","trusted":true,"collapsed":false,"execution":{"iopub.status.busy":"2024-11-18T17:12:10.652086Z","iopub.execute_input":"2024-11-18T17:12:10.652366Z","iopub.status.idle":"2024-11-18T17:12:10.655848Z","shell.execute_reply.started":"2024-11-18T17:12:10.652340Z","shell.execute_reply":"2024-11-18T17:12:10.654996Z"},"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df = df.drop(['date_id', 'time_id', 'symbol_id', 'weight','responder_0', 'responder_1', 'responder_2', 'responder_3', 'responder_4', 'responder_5', 'responder_7', 'responder_8', 'partition_id'])","metadata":{"_uuid":"be722408-3f89-4fad-940a-955d9d460e6e","_cell_guid":"8e828cae-d0a5-457f-b54a-337979f4a264","trusted":true,"collapsed":false,"execution":{"iopub.status.busy":"2024-11-18T17:12:10.656946Z","iopub.execute_input":"2024-11-18T17:12:10.657178Z","iopub.status.idle":"2024-11-18T17:12:10.683485Z","shell.execute_reply.started":"2024-11-18T17:12:10.657155Z","shell.execute_reply":"2024-11-18T17:12:10.682540Z"},"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.null_count()","metadata":{"_uuid":"39d43374-0ebc-4a0a-9944-2a6f20dc99c9","_cell_guid":"5b6f5c6a-d8a7-41c9-9789-32afa2d98823","trusted":true,"collapsed":false,"execution":{"iopub.status.busy":"2024-11-18T17:12:10.684848Z","iopub.execute_input":"2024-11-18T17:12:10.685226Z","iopub.status.idle":"2024-11-18T17:12:10.704220Z","shell.execute_reply.started":"2024-11-18T17:12:10.685190Z","shell.execute_reply":"2024-11-18T17:12:10.703343Z"},"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Drop rows with null values\ndf = df.drop_nulls()\ndf","metadata":{"_uuid":"6db7e7e3-50ed-4a05-aa0d-a58863c4a9b9","_cell_guid":"29d2f63e-fd8d-4ac5-805c-82dc9bae3cd4","trusted":true,"collapsed":false,"execution":{"iopub.status.busy":"2024-11-18T17:12:10.705063Z","iopub.execute_input":"2024-11-18T17:12:10.705315Z","iopub.status.idle":"2024-11-18T17:12:14.569143Z","shell.execute_reply.started":"2024-11-18T17:12:10.705276Z","shell.execute_reply":"2024-11-18T17:12:14.568312Z"},"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model = LGBMRegressor(\n    device='gpu',         # Use GPU for training\n    boosting_type='gbdt', # Gradient Boosting Decision Tree\n    objective='regression',\n    metric='rmse',\n    n_estimators=100,\n    learning_rate=0.1,\n    num_leaves=31,\n    max_depth=-1,\n    force_col_wise=True\n)","metadata":{"_uuid":"4258944e-0b43-48f4-95f4-45bb586ea870","_cell_guid":"3b215f8a-35fc-43e3-86c7-8d8886bc87bb","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2024-11-18T17:12:14.570542Z","iopub.execute_input":"2024-11-18T17:12:14.570930Z","iopub.status.idle":"2024-11-18T17:12:14.575549Z","shell.execute_reply.started":"2024-11-18T17:12:14.570892Z","shell.execute_reply":"2024-11-18T17:12:14.574705Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# train_in_chunks function","metadata":{"_uuid":"2a542d70-e373-45b1-affe-46b63158bf4b","_cell_guid":"30f1607c-c28f-4c8d-8c33-07585ea955fd","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false}}},{"cell_type":"code","source":"\n\ndef train_in_chunks(df, chunksize=100000, test_size=0.2):\n    # Split the data into training and test set once\n    X = df.select([col for col in df.columns if col != \"responder_6\"])\n    y = df[\"responder_6\"]\n    X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=test_size)\n\n    # Initialize progress bar with tqdm\n    num_chunks = len(df) // chunksize + 1\n\n    # Iterate over chunks and train model incrementally\n    for i in tqdm(range(num_chunks), desc=\"Processing Chunks\"):\n        start_idx = i * chunksize\n        end_idx = min((i + 1) * chunksize, len(df))\n        \n        df_chunk = df[start_idx:end_idx]\n\n        # Separate features and target variable\n        X_chunk = df_chunk.select([col for col in df_chunk.columns if col != \"responder_6\"])\n        y_chunk = df_chunk[\"responder_6\"]\n\n        # Fit model incrementally on the current chunk\n        model.fit(X_chunk.to_pandas(), y_chunk.to_pandas(), \n                  eval_set=[(X_test.to_pandas(), y_test.to_pandas())], \n                  eval_metric='l2')\n    \n    # After processing all chunks, make predictions on the test set\n    y_pred = model.predict(X_test.to_pandas())\n\n    # Calculate Mean Squared Error (MSE)\n    mse = mean_squared_error(y_test.to_pandas(), y_pred)\n    # Optionally, calculate R²\n    r2 = r2_score(y_test.to_pandas(), y_pred)\n    \n    print(f\"Mean Squared Error (MSE): {mse}\")\n    print(f\"R-squared (R²): {r2}\")\n\n\n    return model, r2, mse","metadata":{"_uuid":"6c98258e-a44b-4965-8b82-48ea9569b3c1","_cell_guid":"c03db4b6-c518-4032-8865-4235a23eee48","trusted":true,"collapsed":false,"execution":{"iopub.status.busy":"2024-11-18T17:12:14.576745Z","iopub.execute_input":"2024-11-18T17:12:14.577311Z","iopub.status.idle":"2024-11-18T17:12:14.588546Z","shell.execute_reply.started":"2024-11-18T17:12:14.577272Z","shell.execute_reply":"2024-11-18T17:12:14.587845Z"},"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Train the model on your large Parquet dataset and get metrics\nmodel, r2, mse = train_in_chunks(df)\n\nprint(f\"Final R-squared (R²): {r2}\")\nprint(f\"Final Mean Squared Error: {mse}\")","metadata":{"_uuid":"68909f52-b81e-4ef0-b500-918c212e3655","_cell_guid":"63878c06-e916-4c28-b704-112afa7ee4cb","trusted":true,"collapsed":false,"execution":{"iopub.status.busy":"2024-11-18T17:12:28.503260Z","iopub.execute_input":"2024-11-18T17:12:28.503563Z","iopub.status.idle":"2024-11-18T17:15:19.653623Z","shell.execute_reply.started":"2024-11-18T17:12:28.503537Z","shell.execute_reply":"2024-11-18T17:15:19.652211Z"},"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# based on Demo Submission","metadata":{"_uuid":"b0abd6dc-370f-4c98-b3f4-55d8dba66678","_cell_guid":"d2318586-0f17-464e-b768-9a79d62eb5b0","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false}}},{"cell_type":"code","source":"# Assuming `model` is your trained model\n# Assuming features required by the model are named 'feature_00', 'feature_01', etc.\ndef predict(test: pl.DataFrame, lags: pl.DataFrame | None) -> pl.DataFrame | pd.DataFrame:\n    \"\"\"Make a prediction.\"\"\"\n    global lags_\n    if lags is not None:\n        lags_ = lags\n    # Extract the features for the model input\n    feature_columns = [col for col in test.columns if col.startswith(\"feature_\")]\n    features = test.select(feature_columns).to_numpy()  # Convert to numpy array for model input\n    features = np.nan_to_num(features, nan=0.0, posinf=0.0, neginf=0.0)\n    # # Generate predictions using the model\n    responder_6_predictions = model.predict(features)\n    # print(responder_6_predictions)    \n    #responder_6_predictions = model_predictions[:, 6]  # Assuming responder_6 is at index 6\n    # Create a new Polars DataFrame with row_id and responder_6 predictions\n    predictions = test.select(\"row_id\").with_columns(\n        pl.Series(\"responder_6\", responder_6_predictions)\n    )\n    print(predictions)\n    # Ensure the output format and length requirements\n    if isinstance(predictions, pl.DataFrame):\n        assert predictions.columns == ['row_id', 'responder_6']\n    elif isinstance(predictions, pd.DataFrame):\n        assert (predictions.columns == ['row_id', 'responder_6']).all()\n    else:\n        raise TypeError('The predict function must return a DataFrame')\n    \n    assert len(predictions) == len(test)\n    return predictions\ninference_server = kaggle_evaluation.jane_street_inference_server.JSInferenceServer(predict)\nif os.getenv('KAGGLE_IS_COMPETITION_RERUN'):\n    inference_server.serve()\nelse:\n    inference_server.run_local_gateway(\n        (\n            '/kaggle/input/jane-street-real-time-market-data-forecasting/test.parquet',\n            '/kaggle/input/jane-street-real-time-market-data-forecasting/lags.parquet',\n        )\n    )","metadata":{"_uuid":"ced6764d-c841-4a58-96b2-2e95b591129c","_cell_guid":"8787a504-6c61-4fcd-9693-1b7f33524b6e","trusted":true,"collapsed":false,"execution":{"iopub.status.busy":"2024-11-18T17:15:24.429376Z","iopub.execute_input":"2024-11-18T17:15:24.429703Z","iopub.status.idle":"2024-11-18T17:15:24.928336Z","shell.execute_reply.started":"2024-11-18T17:15:24.429675Z","shell.execute_reply":"2024-11-18T17:15:24.927193Z"},"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"_uuid":"b140e74a-5ed5-4775-bd72-ce5d93d502d6","_cell_guid":"57dd9af6-25a9-4613-a731-561fdf61fa66","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null}]}