{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84493,"databundleVersionId":9871156,"sourceType":"competition"}],"dockerImageVersionId":30786,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-10-17T19:50:44.934527Z","iopub.execute_input":"2024-10-17T19:50:44.935146Z","iopub.status.idle":"2024-10-17T19:50:44.974862Z","shell.execute_reply.started":"2024-10-17T19:50:44.935094Z","shell.execute_reply":"2024-10-17T19:50:44.973524Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data_df = pd.read_parquet(\"/kaggle/input/jane-street-real-time-market-data-forecasting/train.parquet/partition_id=2/part-0.parquet\")\ndata_df","metadata":{"execution":{"iopub.status.busy":"2024-10-17T19:50:44.977003Z","iopub.execute_input":"2024-10-17T19:50:44.977406Z","iopub.status.idle":"2024-10-17T19:50:48.437383Z","shell.execute_reply.started":"2024-10-17T19:50:44.977363Z","shell.execute_reply":"2024-10-17T19:50:48.435815Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import StandardScaler\nimport xgboost as xgb\nfrom sklearn.metrics import mean_squared_error\nimport matplotlib.pyplot as plt","metadata":{"execution":{"iopub.status.busy":"2024-10-17T19:50:48.438898Z","iopub.execute_input":"2024-10-17T19:50:48.439422Z","iopub.status.idle":"2024-10-17T19:50:48.445515Z","shell.execute_reply.started":"2024-10-17T19:50:48.439376Z","shell.execute_reply":"2024-10-17T19:50:48.444283Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Prepare the features and target\nfeatures = ['date_id', 'time_id', 'symbol_id', 'weight'] + [f'feature_{i:02d}' for i in range(79)]\nX = data_df[features]\ny = data_df['responder_6']\ny","metadata":{"execution":{"iopub.status.busy":"2024-10-17T19:50:48.448787Z","iopub.execute_input":"2024-10-17T19:50:48.449830Z","iopub.status.idle":"2024-10-17T19:50:48.870678Z","shell.execute_reply.started":"2024-10-17T19:50:48.449767Z","shell.execute_reply":"2024-10-17T19:50:48.869351Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Split the data, ensuring we maintain the chronological order\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, shuffle=False)\n\n# Scale the features\nscaler = StandardScaler()\nX_train_scaled = scaler.fit_transform(X_train)\nX_test_scaled = scaler.transform(X_test)","metadata":{"execution":{"iopub.status.busy":"2024-10-17T19:50:48.872127Z","iopub.execute_input":"2024-10-17T19:50:48.872480Z","iopub.status.idle":"2024-10-17T19:50:56.248682Z","shell.execute_reply.started":"2024-10-17T19:50:48.872443Z","shell.execute_reply":"2024-10-17T19:50:56.247445Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Create and train the XGBoost model\nmodel = xgb.XGBRegressor(n_estimators=100, learning_rate=0.1, random_state=42)\nmodel.fit(X_train_scaled, y_train)\n","metadata":{"execution":{"iopub.status.busy":"2024-10-17T19:50:56.250298Z","iopub.execute_input":"2024-10-17T19:50:56.250820Z","iopub.status.idle":"2024-10-17T19:51:57.561042Z","shell.execute_reply.started":"2024-10-17T19:50:56.250764Z","shell.execute_reply":"2024-10-17T19:51:57.559604Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Make predictions\ny_pred = model.predict(X_test_scaled)","metadata":{"execution":{"iopub.status.busy":"2024-10-17T19:51:57.562805Z","iopub.execute_input":"2024-10-17T19:51:57.563241Z","iopub.status.idle":"2024-10-17T19:51:58.527243Z","shell.execute_reply.started":"2024-10-17T19:51:57.563196Z","shell.execute_reply":"2024-10-17T19:51:58.526205Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Evaluate the model\nmse = mean_squared_error(y_test, y_pred)\nrmse = np.sqrt(mse)\nprint(f\"Root Mean Squared Error: {rmse}\")","metadata":{"execution":{"iopub.status.busy":"2024-10-17T19:51:58.528875Z","iopub.execute_input":"2024-10-17T19:51:58.529729Z","iopub.status.idle":"2024-10-17T19:51:58.539988Z","shell.execute_reply.started":"2024-10-17T19:51:58.529678Z","shell.execute_reply":"2024-10-17T19:51:58.538111Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Feature importance\nfeature_importance = model.feature_importances_\nfeature_names = X.columns\nsorted_idx = np.argsort(feature_importance)\npos = np.arange(sorted_idx.shape[0]) + .5","metadata":{"execution":{"iopub.status.busy":"2024-10-17T19:51:58.542028Z","iopub.execute_input":"2024-10-17T19:51:58.542514Z","iopub.status.idle":"2024-10-17T19:51:58.551993Z","shell.execute_reply.started":"2024-10-17T19:51:58.542470Z","shell.execute_reply":"2024-10-17T19:51:58.551004Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Plot feature importance\nplt.figure(figsize=(12, 6))\nplt.barh(pos, feature_importance[sorted_idx], align='center')\nplt.yticks(pos, feature_names[sorted_idx])\nplt.xlabel('Feature Importance')\nplt.title('XGBoost Feature Importance')\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-17T19:51:58.555747Z","iopub.execute_input":"2024-10-17T19:51:58.556515Z","iopub.status.idle":"2024-10-17T19:51:59.869355Z","shell.execute_reply.started":"2024-10-17T19:51:58.556467Z","shell.execute_reply":"2024-10-17T19:51:59.868055Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Forecast future values\n# We'll forecast for each unique symbol_id in the last date_id and time_id\nlast_date_id = X['date_id'].max()\nlast_time_id = X['time_id'].max()\nunique_symbols = X['symbol_id'].unique()","metadata":{"execution":{"iopub.status.busy":"2024-10-17T19:51:59.871126Z","iopub.execute_input":"2024-10-17T19:51:59.871658Z","iopub.status.idle":"2024-10-17T19:51:59.908795Z","shell.execute_reply.started":"2024-10-17T19:51:59.871581Z","shell.execute_reply":"2024-10-17T19:51:59.906738Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"future_data = []\nfor symbol in unique_symbols:\n    new_row = X[X['symbol_id'] == symbol].iloc[-1].copy()\n    new_row['date_id'] = last_date_id\n    new_row['time_id'] = last_time_id + 1\n    future_data.append(new_row)","metadata":{"execution":{"iopub.status.busy":"2024-10-17T19:51:59.911142Z","iopub.execute_input":"2024-10-17T19:51:59.911729Z","iopub.status.idle":"2024-10-17T19:52:01.874986Z","shell.execute_reply.started":"2024-10-17T19:51:59.911671Z","shell.execute_reply":"2024-10-17T19:52:01.873498Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"future_df = pd.DataFrame(future_data)\nfuture_scaled = scaler.transform(future_df)\n\nforecast = model.predict(future_scaled)\n\n# Clip the forecast between -5 and 5\nforecast = np.clip(forecast, -5, 5)\n\n# Format the forecast results\nforecast_df = pd.DataFrame({\n    'row_id': range(len(forecast)),\n    'responder_6': forecast\n})","metadata":{"execution":{"iopub.status.busy":"2024-10-17T19:52:01.876916Z","iopub.execute_input":"2024-10-17T19:52:01.877523Z","iopub.status.idle":"2024-10-17T19:52:01.896423Z","shell.execute_reply.started":"2024-10-17T19:52:01.877474Z","shell.execute_reply":"2024-10-17T19:52:01.895377Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Display the formatted forecast\nprint(\"\\nForecasted values for responder_6:\")\nprint(forecast_df.to_string(index=False, float_format=lambda x: '{:.1f}'.format(x)))\n","metadata":{"execution":{"iopub.status.busy":"2024-10-17T19:52:01.898060Z","iopub.execute_input":"2024-10-17T19:52:01.898459Z","iopub.status.idle":"2024-10-17T19:52:01.913742Z","shell.execute_reply.started":"2024-10-17T19:52:01.898420Z","shell.execute_reply":"2024-10-17T19:52:01.912360Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Plot the forecast (using the last symbol as an example)\nlast_symbol = X['symbol_id'].iloc[-1]\nsymbol_data = y[X['symbol_id'] == last_symbol]\nplt.figure(figsize=(12, 6))\nplt.plot(symbol_data.index, symbol_data, label='Historical Data')\nplt.scatter(len(symbol_data), forecast[-1], color='red', label='Forecast')\nplt.xlabel('Time')\nplt.ylabel('responder_6')\nplt.title(f'XGBoost Forecast for symbol_id {last_symbol}')\nplt.legend()\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-17T19:52:01.915782Z","iopub.execute_input":"2024-10-17T19:52:01.916313Z","iopub.status.idle":"2024-10-17T19:52:02.949830Z","shell.execute_reply.started":"2024-10-17T19:52:01.916256Z","shell.execute_reply":"2024-10-17T19:52:02.948528Z"},"trusted":true},"outputs":[],"execution_count":null}]}