{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84493,"databundleVersionId":11305158,"sourceType":"competition"}],"dockerImageVersionId":30786,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Dataset Description\n\nThe competition dataset comprises a set of timeseries with 79 features and 9 responders, anonymized but representing real market data. The goal of the competition is to forecast one of these responders, i.e., `responder_6`, for up to six months in the future.\n\nYou must submit to this competition using the provided Python evaluation API, which serves test set data one timestep by timestep. To use the API, follow the example in [this notebook](https://www.kaggle.com/code/ryanholbrook/jane-street-rmf-demo-submission). (Note that this API is different from our legacy timeseries API used in past forecasting competitions.)","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport polars as pl\nfrom matplotlib import pyplot as plt\nfrom matplotlib.ticker import MaxNLocator, FormatStrFormatter, PercentFormatter\nimport seaborn as sns","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2025-06-08T17:37:40.977041Z","iopub.execute_input":"2025-06-08T17:37:40.977632Z","iopub.status.idle":"2025-06-08T17:37:40.984291Z","shell.execute_reply.started":"2025-06-08T17:37:40.977586Z","shell.execute_reply":"2025-06-08T17:37:40.983026Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"ROOT_DIR = \"/kaggle/input/jane-street-real-time-market-data-forecasting\"","metadata":{"execution":{"iopub.status.busy":"2025-06-08T17:37:40.986668Z","iopub.execute_input":"2025-06-08T17:37:40.987152Z","iopub.status.idle":"2025-06-08T17:37:41.000315Z","shell.execute_reply.started":"2025-06-08T17:37:40.987102Z","shell.execute_reply":"2025-06-08T17:37:40.998944Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Features\n\n- features.csv - metadata pertaining to the anonymized features","metadata":{}},{"cell_type":"code","source":"features = pd.read_csv(f\"{ROOT_DIR}/features.csv\")\nfeatures","metadata":{"execution":{"iopub.status.busy":"2025-06-08T17:37:41.015992Z","iopub.execute_input":"2025-06-08T17:37:41.016397Z","iopub.status.idle":"2025-06-08T17:37:41.050251Z","shell.execute_reply.started":"2025-06-08T17:37:41.016360Z","shell.execute_reply":"2025-06-08T17:37:41.048766Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(20, 10))\nplt.imshow(features.iloc[:, 1:].T.values, cmap=\"gray\")\nplt.xlabel(\"feature_00  ~  feature_78\")\nplt.ylabel(\"tag_0  ~  tag_16\")\nplt.yticks(np.arange(17))\nplt.xticks(np.arange(79))\nplt.grid()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2025-06-08T17:37:41.052267Z","iopub.execute_input":"2025-06-08T17:37:41.052689Z","iopub.status.idle":"2025-06-08T17:37:41.790163Z","shell.execute_reply.started":"2025-06-08T17:37:41.052649Z","shell.execute_reply":"2025-06-08T17:37:41.788880Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# corr between feature_XX and feature_YY\nplt.figure(figsize=(10, 10))\nsns.heatmap(features[[ f\"tag_{no}\" for no in range(0,17,1) ] ].T.corr(), square=True, cmap=\"jet\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2025-06-08T17:37:41.792878Z","iopub.execute_input":"2025-06-08T17:37:41.793277Z","iopub.status.idle":"2025-06-08T17:37:42.812855Z","shell.execute_reply.started":"2025-06-08T17:37:41.793238Z","shell.execute_reply":"2025-06-08T17:37:42.811677Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Responders\n\n- responders.csv - metadata pertaining to the anonymized responders","metadata":{}},{"cell_type":"code","source":"responders = pd.read_csv(f\"{ROOT_DIR}/responders.csv\")\nresponders","metadata":{"execution":{"iopub.status.busy":"2025-06-08T17:37:42.814091Z","iopub.execute_input":"2025-06-08T17:37:42.814430Z","iopub.status.idle":"2025-06-08T17:37:42.834654Z","shell.execute_reply.started":"2025-06-08T17:37:42.814395Z","shell.execute_reply":"2025-06-08T17:37:42.833192Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# corr between responder_XX and responder_YY\nsns.heatmap(responders[[ f\"tag_{no}\" for no in range(0,5,1) ] ].T.corr(),  annot=True, square=True, cmap=\"jet\")\nplt.xlabel(\"responder_0  ~  responder_8\")\nplt.ylabel(\"responder_0  ~  responder_8\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2025-06-08T17:37:42.838002Z","iopub.execute_input":"2025-06-08T17:37:42.838473Z","iopub.status.idle":"2025-06-08T17:37:43.347928Z","shell.execute_reply.started":"2025-06-08T17:37:42.838398Z","shell.execute_reply":"2025-06-08T17:37:43.346630Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Sample submission\n\n- **sample_submission.csv** - This file illustrates the format of the predictions your model should make.","metadata":{}},{"cell_type":"code","source":"sub = pd.read_csv(f\"{ROOT_DIR}/sample_submission.csv\")\nprint( f\"sub.shape = {sub.shape}\" )\nsub","metadata":{"execution":{"iopub.status.busy":"2025-06-08T17:37:43.349912Z","iopub.execute_input":"2025-06-08T17:37:43.350373Z","iopub.status.idle":"2025-06-08T17:37:43.371859Z","shell.execute_reply.started":"2025-06-08T17:37:43.350325Z","shell.execute_reply":"2025-06-08T17:37:43.370553Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Train.parquet\n\n- **train.parquet** - The training set, contains historical data and returns. For convenience, the training set has been partitioned into ten parts.\n  - `date_id` and `time_id` - Integer values that are ordinally sorted, providing a chronological structure to the data, although the actual time intervals between `time_id` values may vary.\n  - `symbol_id` - Identifies a unique financial instrument.\n  - `weight` - The weighting used for calculating the scoring function.\n  - `feature_{00...78}` - Anonymized market data.\n  - `responder_{0...8}` - Anonymized responders clipped between -5 and 5. The responder_6 field is what you are trying to predict.\n  \n  \nEach row in the `{train/test}.parquet` dataset corresponds to a unique combination of a symbol (identified by `symbol_id`) and a timestamp (represented by `date_id` and `time_id`). You will be provided with multiple responders, with `responder_6` being the only responder used for scoring. The date_id column is an integer which represents the day of the event, while time_id represents a time ordering. It's important to note that the real time differences between each time_id are not guaranteed to be consistent.\n\nThe `symbol_id` column contains encrypted identifiers. Each `symbol_id` is not guaranteed to appear in all `time_id` and `date_id` combinations. Additionally, new `symbol_id` values may appear in future test sets.","metadata":{}},{"cell_type":"code","source":"!tree /kaggle/input/jane-street-real-time-market-data-forecasting/train.parquet/","metadata":{"execution":{"iopub.status.busy":"2025-06-08T17:37:43.373220Z","iopub.execute_input":"2025-06-08T17:37:43.373594Z","iopub.status.idle":"2025-06-08T17:37:44.814936Z","shell.execute_reply.started":"2025-06-08T17:37:43.373558Z","shell.execute_reply":"2025-06-08T17:37:44.813250Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = (\n    pl.read_parquet(f\"{ROOT_DIR}/train.parquet/partition_id=0/part-0.parquet\")\n)\ntrain.shape","metadata":{"execution":{"iopub.status.busy":"2025-06-08T17:37:44.817728Z","iopub.execute_input":"2025-06-08T17:37:44.818222Z","iopub.status.idle":"2025-06-08T17:37:45.535278Z","shell.execute_reply.started":"2025-06-08T17:37:44.818178Z","shell.execute_reply":"2025-06-08T17:37:45.533948Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.head()","metadata":{"execution":{"iopub.status.busy":"2025-06-08T17:37:45.537028Z","iopub.execute_input":"2025-06-08T17:37:45.537511Z","iopub.status.idle":"2025-06-08T17:37:45.551780Z","shell.execute_reply.started":"2025-06-08T17:37:45.537456Z","shell.execute_reply":"2025-06-08T17:37:45.550279Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(str(train.columns))","metadata":{"execution":{"iopub.status.busy":"2025-06-08T17:37:45.553175Z","iopub.execute_input":"2025-06-08T17:37:45.553549Z","iopub.status.idle":"2025-06-08T17:37:45.565118Z","shell.execute_reply.started":"2025-06-08T17:37:45.553515Z","shell.execute_reply":"2025-06-08T17:37:45.563921Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Missing values","metadata":{}},{"cell_type":"code","source":"for partition_id in range(10):\n    print(f\"> train.parquet/partition_id={partition_id}/part-0.parquet\")\n    train = pl.read_parquet(f\"{ROOT_DIR}/train.parquet/partition_id={partition_id}/part-0.parquet\")\n    supervised_usable = (\n    train\n    .filter(pl.col('responder_6').is_not_null())\n    )\n    \n    missing_count = (\n        supervised_usable\n        .null_count()\n        .transpose(include_header=True,\n                   header_name='feature',\n                   column_names=['null_count'])\n        .sort('null_count', descending=True)\n        .with_columns((pl.col('null_count') / len(supervised_usable)).alias('null_ratio'))\n    )\n    \n    plt.figure(figsize=(6, 20))\n    plt.title(f'Missing values over the {len(supervised_usable)} samples which have a target')\n    plt.barh(np.arange(len(missing_count)), missing_count.get_column('null_ratio'), color='coral', label='missing')\n    plt.barh(np.arange(len(missing_count)), \n             1 - missing_count.get_column('null_ratio'),\n             left=missing_count.get_column('null_ratio'),\n             color='darkseagreen', label='available')\n    plt.yticks(np.arange(len(missing_count)), missing_count.get_column('feature'))\n    plt.gca().xaxis.set_major_formatter(PercentFormatter(xmax=1, decimals=0))\n    plt.xlim(0, 1)\n    plt.legend()\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-08T17:37:45.568593Z","iopub.execute_input":"2025-06-08T17:37:45.569168Z","iopub.status.idle":"2025-06-08T17:38:12.891656Z","shell.execute_reply.started":"2025-06-08T17:37:45.569126Z","shell.execute_reply":"2025-06-08T17:38:12.890469Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"supervised_usable = (\n    train\n    .filter(pl.col('responder_6').is_not_null())\n)\n\nmissing_count = (\n    supervised_usable\n    .null_count()\n    .transpose(include_header=True,\n               header_name='feature',\n               column_names=['null_count'])\n    .sort('null_count', descending=True)\n    .with_columns((pl.col('null_count') / len(supervised_usable)).alias('null_ratio'))\n)\n\nplt.figure(figsize=(6, 20))\nplt.title(f'Missing values over the {len(supervised_usable)} samples which have a target')\nplt.barh(np.arange(len(missing_count)), missing_count.get_column('null_ratio'), color='coral', label='missing')\nplt.barh(np.arange(len(missing_count)), \n         1 - missing_count.get_column('null_ratio'),\n         left=missing_count.get_column('null_ratio'),\n         color='darkseagreen', label='available')\nplt.yticks(np.arange(len(missing_count)), missing_count.get_column('feature'))\nplt.gca().xaxis.set_major_formatter(PercentFormatter(xmax=1, decimals=0))\nplt.xlim(0, 1)\nplt.legend()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2025-06-08T17:38:12.893042Z","iopub.execute_input":"2025-06-08T17:38:12.893359Z","iopub.status.idle":"2025-06-08T17:38:14.072928Z","shell.execute_reply.started":"2025-06-08T17:38:12.893328Z","shell.execute_reply":"2025-06-08T17:38:14.071727Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## feature_00-78","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(15, 15))\nsns.heatmap(train[[ f\"feature_{target:02d}\" for target in range(79)]].corr(), square=True, cmap=\"jet\")\nplt.xlabel(\"feature_00  ~  feature_78\")\nplt.ylabel(\"feature_00  ~  feature_78\")\nplt.grid()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2025-06-08T17:38:14.074909Z","iopub.execute_input":"2025-06-08T17:38:14.075286Z","iopub.status.idle":"2025-06-08T17:38:20.552624Z","shell.execute_reply.started":"2025-06-08T17:38:14.075248Z","shell.execute_reply":"2025-06-08T17:38:20.551355Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## responder_0 - 8","metadata":{}},{"cell_type":"code","source":"for target in range(9):\n    col = f\"responder_{target}\"\n    mean_, sgm_ = train[col].mean(), np.sqrt(train[col].var())\n    min_, max_ = train[col].min(), train[col].max()\n    print(\"- \" * 30)\n    print( f\"column = {col}\" )\n    print( f\" - mean  : {mean_:.4f}\",  )\n    print( f\" - sigma : {sgm_:.4f}\",  )\n    print( f\" - min  : {min_:.4f}\",  )\n    print( f\" - max  : {max_:.4f}\",  )\n    \n    plt.hist(train[col], bins=20)\n    plt.xlabel(col)\n    plt.ylabel(\"frequency / records\")\n    #plt.yscale(\"log\")\n    plt.grid()\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2025-06-08T17:38:20.554311Z","iopub.execute_input":"2025-06-08T17:38:20.554799Z","iopub.status.idle":"2025-06-08T17:38:24.059566Z","shell.execute_reply.started":"2025-06-08T17:38:20.554737Z","shell.execute_reply":"2025-06-08T17:38:24.058317Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(8, 8))\nsns.heatmap(train[[ f\"responder_{target}\" for target in range(9)]].corr(),  annot=True, square=True, cmap=\"jet\")\nplt.xlabel(\"responder_0  ~  responder_8\")\nplt.ylabel(\"responder_0  ~  responder_8\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2025-06-08T17:38:24.061104Z","iopub.execute_input":"2025-06-08T17:38:24.061501Z","iopub.status.idle":"2025-06-08T17:38:25.007107Z","shell.execute_reply.started":"2025-06-08T17:38:24.061428Z","shell.execute_reply":"2025-06-08T17:38:25.006010Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## symbol_id","metadata":{}},{"cell_type":"code","source":"for partition_id in range(10):\n    print(f\"> train.parquet/partition_id={partition_id}/part-0.parquet\")\n    train_data = pl.read_parquet(f\"{ROOT_DIR}/train.parquet/partition_id={partition_id}/part-0.parquet\")\n\n    print( f\"symbol_id: \", train_data[\"symbol_id\"].min(), \"-\", train_data[\"symbol_id\"].max())\n    bins = train_data[\"symbol_id\"].max() - train_data[\"symbol_id\"].min() + 1\n    plt.hist(train_data[\"symbol_id\"], bins=bins)\n    plt.xlabel(\"symbol_id\")\n    plt.ylabel(\"frequency / records\")\n    plt.grid()\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2025-06-08T17:38:25.008571Z","iopub.execute_input":"2025-06-08T17:38:25.008955Z","iopub.status.idle":"2025-06-08T17:38:43.302958Z","shell.execute_reply.started":"2025-06-08T17:38:25.008918Z","shell.execute_reply":"2025-06-08T17:38:43.301391Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## date_id","metadata":{}},{"cell_type":"code","source":"for partition_id in range(10):\n    print(f\"> train.parquet/partition_id={partition_id}/part-0.parquet\")\n    train_data = pl.read_parquet(f\"{ROOT_DIR}/train.parquet/partition_id={partition_id}/part-0.parquet\")\n\n    print( f\"date_id: \", train_data[\"date_id\"].min(), \"-\", train_data[\"date_id\"].max())\n    bins = train_data[\"date_id\"].max() - train_data[\"date_id\"].min() + 1\n    plt.hist(train_data[\"date_id\"], bins=bins)\n    plt.xlabel(\"date_id\")\n    plt.ylabel(\"frequency / records\")\n    plt.grid()\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-08T17:38:43.304678Z","iopub.execute_input":"2025-06-08T17:38:43.305192Z","iopub.status.idle":"2025-06-08T17:39:03.369818Z","shell.execute_reply.started":"2025-06-08T17:38:43.305134Z","shell.execute_reply":"2025-06-08T17:39:03.368615Z"}},"outputs":[],"execution_count":null}]}