{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84493,"databundleVersionId":9871156,"sourceType":"competition"}],"dockerImageVersionId":30786,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Dataset Description\n\nThe competition dataset comprises a set of timeseries with 79 features and 9 responders, anonymized but representing real market data. The goal of the competition is to forecast one of these responders, i.e., `responder_6`, for up to six months in the future.\n\nYou must submit to this competition using the provided Python evaluation API, which serves test set data one timestep by timestep. To use the API, follow the example in [this notebook](https://www.kaggle.com/code/ryanholbrook/jane-street-rmf-demo-submission). (Note that this API is different from our legacy timeseries API used in past forecasting competitions.)","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport polars as pl\nfrom matplotlib import pyplot as plt\nfrom matplotlib.ticker import MaxNLocator, FormatStrFormatter, PercentFormatter\nimport seaborn as sns","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-12-14T13:00:51.692081Z","iopub.execute_input":"2024-12-14T13:00:51.692512Z","iopub.status.idle":"2024-12-14T13:00:55.304289Z","shell.execute_reply.started":"2024-12-14T13:00:51.692473Z","shell.execute_reply":"2024-12-14T13:00:55.303249Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"ROOT_DIR = \"/kaggle/input/jane-street-real-time-market-data-forecasting\"","metadata":{"execution":{"iopub.status.busy":"2024-12-14T13:00:55.306399Z","iopub.execute_input":"2024-12-14T13:00:55.307046Z","iopub.status.idle":"2024-12-14T13:00:55.312145Z","shell.execute_reply.started":"2024-12-14T13:00:55.306994Z","shell.execute_reply":"2024-12-14T13:00:55.311054Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Features\n\n- features.csv - metadata pertaining to the anonymized features","metadata":{}},{"cell_type":"code","source":"features = pd.read_csv(f\"{ROOT_DIR}/features.csv\")\nfeatures","metadata":{"execution":{"iopub.status.busy":"2024-12-14T13:00:55.313394Z","iopub.execute_input":"2024-12-14T13:00:55.313711Z","iopub.status.idle":"2024-12-14T13:00:55.374349Z","shell.execute_reply.started":"2024-12-14T13:00:55.313680Z","shell.execute_reply":"2024-12-14T13:00:55.373267Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(20, 10))\nplt.imshow(features.iloc[:, 1:].T.values, cmap=\"gray\")\nplt.xlabel(\"feature_00  ~  feature_78\")\nplt.ylabel(\"tag_0  ~  tag_16\")\nplt.yticks(np.arange(17))\nplt.xticks(np.arange(79))\nplt.grid()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-12-14T13:00:55.376855Z","iopub.execute_input":"2024-12-14T13:00:55.377203Z","iopub.status.idle":"2024-12-14T13:00:56.118105Z","shell.execute_reply.started":"2024-12-14T13:00:55.377173Z","shell.execute_reply":"2024-12-14T13:00:56.117030Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# corr between feature_XX and feature_YY\nplt.figure(figsize=(10, 10))\nsns.heatmap(features[[ f\"tag_{no}\" for no in range(0,17,1) ] ].T.corr(), square=True, cmap=\"jet\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-12-14T13:00:56.119484Z","iopub.execute_input":"2024-12-14T13:00:56.119872Z","iopub.status.idle":"2024-12-14T13:00:56.827826Z","shell.execute_reply.started":"2024-12-14T13:00:56.119840Z","shell.execute_reply":"2024-12-14T13:00:56.826749Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Responders\n\n- responders.csv - metadata pertaining to the anonymized responders","metadata":{}},{"cell_type":"code","source":"responders = pd.read_csv(f\"{ROOT_DIR}/responders.csv\")\nresponders","metadata":{"execution":{"iopub.status.busy":"2024-12-14T13:00:56.829140Z","iopub.execute_input":"2024-12-14T13:00:56.829620Z","iopub.status.idle":"2024-12-14T13:00:56.847581Z","shell.execute_reply.started":"2024-12-14T13:00:56.829586Z","shell.execute_reply":"2024-12-14T13:00:56.846460Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# corr between responder_XX and responder_YY\nsns.heatmap(responders[[ f\"tag_{no}\" for no in range(0,5,1) ] ].T.corr(),  annot=True, square=True, cmap=\"jet\")\nplt.xlabel(\"responder_0  ~  responder_8\")\nplt.ylabel(\"responder_0  ~  responder_8\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-12-14T13:00:56.849146Z","iopub.execute_input":"2024-12-14T13:00:56.849635Z","iopub.status.idle":"2024-12-14T13:00:57.432141Z","shell.execute_reply.started":"2024-12-14T13:00:56.849586Z","shell.execute_reply":"2024-12-14T13:00:57.431167Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Sample submission\n\n- **sample_submission.csv** - This file illustrates the format of the predictions your model should make.","metadata":{}},{"cell_type":"code","source":"sub = pd.read_csv(f\"{ROOT_DIR}/sample_submission.csv\")\nprint( f\"sub.shape = {sub.shape}\" )\nsub","metadata":{"execution":{"iopub.status.busy":"2024-12-14T13:00:57.433614Z","iopub.execute_input":"2024-12-14T13:00:57.434100Z","iopub.status.idle":"2024-12-14T13:00:57.454354Z","shell.execute_reply.started":"2024-12-14T13:00:57.434051Z","shell.execute_reply":"2024-12-14T13:00:57.453295Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Train.parquet\n\n- **train.parquet** - The training set, contains historical data and returns. For convenience, the training set has been partitioned into ten parts.\n  - `date_id` and `time_id` - Integer values that are ordinally sorted, providing a chronological structure to the data, although the actual time intervals between `time_id` values may vary.\n  - `symbol_id` - Identifies a unique financial instrument.\n  - `weight` - The weighting used for calculating the scoring function.\n  - `feature_{00...78}` - Anonymized market data.\n  - `responder_{0...8}` - Anonymized responders clipped between -5 and 5. The responder_6 field is what you are trying to predict.\n  \n  \nEach row in the `{train/test}.parquet` dataset corresponds to a unique combination of a symbol (identified by `symbol_id`) and a timestamp (represented by `date_id` and `time_id`). You will be provided with multiple responders, with `responder_6` being the only responder used for scoring. The date_id column is an integer which represents the day of the event, while time_id represents a time ordering. It's important to note that the real time differences between each time_id are not guaranteed to be consistent.\n\nThe `symbol_id` column contains encrypted identifiers. Each `symbol_id` is not guaranteed to appear in all `time_id` and `date_id` combinations. Additionally, new `symbol_id` values may appear in future test sets.","metadata":{}},{"cell_type":"code","source":"!tree /kaggle/input/jane-street-real-time-market-data-forecasting/train.parquet/","metadata":{"execution":{"iopub.status.busy":"2024-12-14T13:00:57.455771Z","iopub.execute_input":"2024-12-14T13:00:57.456447Z","iopub.status.idle":"2024-12-14T13:00:58.663215Z","shell.execute_reply.started":"2024-12-14T13:00:57.456398Z","shell.execute_reply":"2024-12-14T13:00:58.661850Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = (\n    pl.read_parquet(f\"{ROOT_DIR}/train.parquet/partition_id=0/part-0.parquet\")\n)\ntrain.shape","metadata":{"execution":{"iopub.status.busy":"2024-12-14T13:00:58.667836Z","iopub.execute_input":"2024-12-14T13:00:58.668264Z","iopub.status.idle":"2024-12-14T13:01:01.018825Z","shell.execute_reply.started":"2024-12-14T13:00:58.668218Z","shell.execute_reply":"2024-12-14T13:01:01.017741Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.head()","metadata":{"execution":{"iopub.status.busy":"2024-12-14T13:01:01.020018Z","iopub.execute_input":"2024-12-14T13:01:01.020328Z","iopub.status.idle":"2024-12-14T13:01:01.047381Z","shell.execute_reply.started":"2024-12-14T13:01:01.020299Z","shell.execute_reply":"2024-12-14T13:01:01.046192Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(str(train.columns))","metadata":{"execution":{"iopub.status.busy":"2024-12-14T13:01:01.048560Z","iopub.execute_input":"2024-12-14T13:01:01.049007Z","iopub.status.idle":"2024-12-14T13:01:01.060456Z","shell.execute_reply.started":"2024-12-14T13:01:01.048951Z","shell.execute_reply":"2024-12-14T13:01:01.059241Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.schema","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T13:01:01.061838Z","iopub.execute_input":"2024-12-14T13:01:01.062233Z","iopub.status.idle":"2024-12-14T13:01:01.082040Z","shell.execute_reply.started":"2024-12-14T13:01:01.062199Z","shell.execute_reply":"2024-12-14T13:01:01.080766Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Missing values","metadata":{}},{"cell_type":"code","source":"# only look at the data where responder_6 is not null\nsupervised_usable = (\n    train\n    .filter(pl.col('responder_6').is_not_null())\n)\n\nmissing_count = (\n    supervised_usable\n    .null_count()  # Counts null values in each column\n    .transpose(\n        include_header=True,\n        header_name='feature',\n        column_names=['null_count']\n    )  # Transposes the DataFrame, making columns into rows\n    .sort('null_count', descending=True)  # Sorts by number of nulls (highest to lowest)\n    .with_columns(\n        (pl.col('null_count') / len(supervised_usable)).alias('null_ratio')\n    )  # Adds a new column showing the ratio of nulls\n)\n\nplt.figure(figsize=(6, 20))\nplt.title(f'Missing values over the {len(supervised_usable)} samples which have a target')\nplt.barh(np.arange(len(missing_count)), missing_count.get_column('null_ratio'), color='coral', label='missing')\nplt.barh(np.arange(len(missing_count)), \n         1 - missing_count.get_column('null_ratio'),\n         left=missing_count.get_column('null_ratio'),\n         color='darkseagreen', label='available')\nplt.yticks(np.arange(len(missing_count)), missing_count.get_column('feature'))\nplt.gca().xaxis.set_major_formatter(PercentFormatter(xmax=1, decimals=0))\nplt.xlim(0, 1)\nplt.legend()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-12-14T13:01:01.083385Z","iopub.execute_input":"2024-12-14T13:01:01.083691Z","iopub.status.idle":"2024-12-14T13:01:02.358577Z","shell.execute_reply.started":"2024-12-14T13:01:01.083663Z","shell.execute_reply":"2024-12-14T13:01:02.357496Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## feature_00-78","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(15, 15))\nsns.heatmap(train[[ f\"feature_{target:02d}\" for target in range(79)]].corr(), square=True, cmap=\"jet\")\nplt.xlabel(\"feature_00  ~  feature_78\")\nplt.ylabel(\"feature_00  ~  feature_78\")\nplt.grid()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-12-14T13:01:02.359735Z","iopub.execute_input":"2024-12-14T13:01:02.360074Z","iopub.status.idle":"2024-12-14T13:01:04.980578Z","shell.execute_reply.started":"2024-12-14T13:01:02.360042Z","shell.execute_reply":"2024-12-14T13:01:04.979453Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## responder_0 - 8","metadata":{}},{"cell_type":"code","source":"for target in range(9):\n    col = f\"responder_{target}\"\n    mean_, sgm_ = train[col].mean(), np.sqrt(train[col].var())\n    min_, max_ = train[col].min(), train[col].max()\n    print(\"- \" * 30)\n    print( f\"column = {col}\" )\n    print( f\" - mean  : {mean_:.4f}\",  )\n    print( f\" - sigma : {sgm_:.4f}\",  )\n    print( f\" - min  : {min_:.4f}\",  )\n    print( f\" - max  : {max_:.4f}\",  )\n    \n    plt.hist(train[col], bins=20)\n    plt.xlabel(col)\n    plt.ylabel(\"frequency / records\")\n    #plt.yscale(\"log\")\n    plt.grid()\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2024-12-14T13:01:04.982267Z","iopub.execute_input":"2024-12-14T13:01:04.983070Z","iopub.status.idle":"2024-12-14T13:01:07.055906Z","shell.execute_reply.started":"2024-12-14T13:01:04.983019Z","shell.execute_reply":"2024-12-14T13:01:07.054912Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(8, 8))\nsns.heatmap(train[[ f\"responder_{target}\" for target in range(9)]].corr(),  annot=True, square=True, cmap=\"jet\")\nplt.xlabel(\"responder_0  ~  responder_8\")\nplt.ylabel(\"responder_0  ~  responder_8\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-12-14T13:01:07.057438Z","iopub.execute_input":"2024-12-14T13:01:07.057873Z","iopub.status.idle":"2024-12-14T13:01:07.597714Z","shell.execute_reply.started":"2024-12-14T13:01:07.057840Z","shell.execute_reply":"2024-12-14T13:01:07.596568Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## symbol_id","metadata":{}},{"cell_type":"code","source":"for partition_id in range(10):\n    print(f\"> train.parquet/partition_id={partition_id}/part-0.parquet\")\n    train_data = pl.read_parquet(f\"{ROOT_DIR}/train.parquet/partition_id={partition_id}/part-0.parquet\")\n\n    print( f\"symbol_id: \", train_data[\"symbol_id\"].min(), \"-\", train_data[\"symbol_id\"].max())\n    bins = train_data[\"symbol_id\"].max() - train_data[\"symbol_id\"].min() + 1\n    plt.hist(train_data[\"symbol_id\"], bins=bins)\n    plt.xlabel(\"symbol_id\")\n    plt.ylabel(\"frequency / records\")\n    plt.grid()\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2024-12-14T13:01:07.599787Z","iopub.execute_input":"2024-12-14T13:01:07.600238Z","iopub.status.idle":"2024-12-14T13:01:57.410112Z","shell.execute_reply.started":"2024-12-14T13:01:07.600190Z","shell.execute_reply":"2024-12-14T13:01:57.409018Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## date_id","metadata":{}},{"cell_type":"code","source":"for partition_id in range(10):\n    print(f\"> train.parquet/partition_id={partition_id}/part-0.parquet\")\n    train_data = pl.read_parquet(f\"{ROOT_DIR}/train.parquet/partition_id={partition_id}/part-0.parquet\")\n\n    print( f\"date_id: \", train_data[\"date_id\"].min(), \"-\", train_data[\"date_id\"].max())\n    bins = train_data[\"date_id\"].max() - train_data[\"date_id\"].min() + 1\n    plt.hist(train_data[\"date_id\"], bins=bins)\n    plt.xlabel(\"date_id\")\n    plt.ylabel(\"frequency / records\")\n    plt.grid()\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T13:01:57.412022Z","iopub.execute_input":"2024-12-14T13:01:57.412364Z","iopub.status.idle":"2024-12-14T13:02:17.334189Z","shell.execute_reply.started":"2024-12-14T13:01:57.412331Z","shell.execute_reply":"2024-12-14T13:02:17.333172Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Test.parquet\n\n- **test.parquet** - A mock test set which represents the structure of the unseen test set. This example set demonstrates a single batch served by the evaluation API, that is, data from a single `date_id, time_id` pair. The test set contains columns including `date_id`, `time_id`, `symbol_id`, `weight` and `feature_{00...78}`. You will not be directly using the test set or sample submission in this competition, as the evaluation API will get/set the test set and predictions.","metadata":{}},{"cell_type":"code","source":"!tree {ROOT_DIR}/test.parquet/","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T13:02:17.335716Z","iopub.execute_input":"2024-12-14T13:02:17.336648Z","iopub.status.idle":"2024-12-14T13:02:18.710759Z","shell.execute_reply.started":"2024-12-14T13:02:17.336598Z","shell.execute_reply":"2024-12-14T13:02:18.709343Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test = (\n    pl.read_parquet(f\"{ROOT_DIR}/test.parquet/date_id=0/part-0.parquet\")\n)\ntest.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T13:02:18.713266Z","iopub.execute_input":"2024-12-14T13:02:18.713677Z","iopub.status.idle":"2024-12-14T13:02:18.734324Z","shell.execute_reply.started":"2024-12-14T13:02:18.713636Z","shell.execute_reply":"2024-12-14T13:02:18.732762Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T13:02:18.736121Z","iopub.execute_input":"2024-12-14T13:02:18.736566Z","iopub.status.idle":"2024-12-14T13:02:18.754993Z","shell.execute_reply.started":"2024-12-14T13:02:18.736524Z","shell.execute_reply":"2024-12-14T13:02:18.753636Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Missing values","metadata":{}},{"cell_type":"code","source":"supervised_usable = (\n    test\n)\n\nmissing_count = (\n    supervised_usable\n    .null_count()\n    .transpose(include_header=True,\n               header_name='feature',\n               column_names=['null_count'])\n    .sort('null_count', descending=True)\n    .with_columns((pl.col('null_count') / len(supervised_usable)).alias('null_ratio'))\n)\n\nplt.figure(figsize=(6, 20))\nplt.title(f'Missing values over the {len(supervised_usable)} samples which have a target')\nplt.barh(np.arange(len(missing_count)), missing_count.get_column('null_ratio'), color='coral', label='missing')\nplt.barh(np.arange(len(missing_count)), \n         1 - missing_count.get_column('null_ratio'),\n         left=missing_count.get_column('null_ratio'),\n         color='darkseagreen', label='available')\nplt.yticks(np.arange(len(missing_count)), missing_count.get_column('feature'))\nplt.gca().xaxis.set_major_formatter(PercentFormatter(xmax=1, decimals=0))\nplt.xlim(0, 1)\nplt.legend()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T13:02:18.756398Z","iopub.execute_input":"2024-12-14T13:02:18.756751Z","iopub.status.idle":"2024-12-14T13:02:19.894537Z","shell.execute_reply.started":"2024-12-14T13:02:18.756719Z","shell.execute_reply":"2024-12-14T13:02:19.893426Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# lags.parquet\n\n- `lags.parquet` - Values of `responder_{0...8}` lagged by one `date_id`. The evaluation API serves the entirety of the lagged responders for a `date_id` on that date_id's first `time_id`. In other words, all of the previous date's responders will be served at the first time step of the succeeding date.","metadata":{}},{"cell_type":"code","source":"!tree {ROOT_DIR}/lags.parquet","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T13:02:19.895875Z","iopub.execute_input":"2024-12-14T13:02:19.896260Z","iopub.status.idle":"2024-12-14T13:02:21.263676Z","shell.execute_reply.started":"2024-12-14T13:02:19.896227Z","shell.execute_reply":"2024-12-14T13:02:21.262623Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"lags = (\n    pl.read_parquet(f\"{ROOT_DIR}/lags.parquet/date_id=0/part-0.parquet\")\n)\nlags.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T13:02:21.265231Z","iopub.execute_input":"2024-12-14T13:02:21.265565Z","iopub.status.idle":"2024-12-14T13:02:21.277568Z","shell.execute_reply.started":"2024-12-14T13:02:21.265534Z","shell.execute_reply":"2024-12-14T13:02:21.276325Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"lags.columns","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T13:02:21.279144Z","iopub.execute_input":"2024-12-14T13:02:21.279488Z","iopub.status.idle":"2024-12-14T13:02:21.288329Z","shell.execute_reply.started":"2024-12-14T13:02:21.279456Z","shell.execute_reply":"2024-12-14T13:02:21.287217Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"lags","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T13:02:21.289560Z","iopub.execute_input":"2024-12-14T13:02:21.289842Z","iopub.status.idle":"2024-12-14T13:02:21.305487Z","shell.execute_reply.started":"2024-12-14T13:02:21.289815Z","shell.execute_reply":"2024-12-14T13:02:21.304298Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.plot(lags[\"responder_6_lag_1\"])\nplt.grid()\nplt.xlabel(\"symbol_id\")\nplt.ylabel(\"responder_6_lag_1\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T13:02:21.311234Z","iopub.execute_input":"2024-12-14T13:02:21.312075Z","iopub.status.idle":"2024-12-14T13:02:21.543979Z","shell.execute_reply.started":"2024-12-14T13:02:21.312035Z","shell.execute_reply":"2024-12-14T13:02:21.542979Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Missing value (Null) analysis","metadata":{}},{"cell_type":"code","source":"train_all = pl.scan_parquet(\"/kaggle/input/jane-street-real-time-market-data-forecasting/train.parquet\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T13:03:54.146032Z","iopub.execute_input":"2024-12-14T13:03:54.146468Z","iopub.status.idle":"2024-12-14T13:03:54.151986Z","shell.execute_reply.started":"2024-12-14T13:03:54.146434Z","shell.execute_reply":"2024-12-14T13:03:54.150773Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Count null(NaN) for eatch columns (group by date_id)\nnull_count_per_date_id = train_all.group_by(\"date_id\").agg(pl.all().null_count()).collect()\nnull_count_per_date_id","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T13:03:56.502408Z","iopub.execute_input":"2024-12-14T13:03:56.502812Z","iopub.status.idle":"2024-12-14T13:04:43.960593Z","shell.execute_reply.started":"2024-12-14T13:03:56.502776Z","shell.execute_reply":"2024-12-14T13:04:43.959478Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Counter number of records group by date_id\nrecords_date_id = train_all.group_by(\"date_id\").agg(pl.count().alias(\"num_records\")).collect()\nrecords_date_id","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T13:05:16.359244Z","iopub.execute_input":"2024-12-14T13:05:16.359612Z","iopub.status.idle":"2024-12-14T13:05:17.147256Z","shell.execute_reply.started":"2024-12-14T13:05:16.359583Z","shell.execute_reply":"2024-12-14T13:05:17.146126Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Count null(NaN) for all features columns (group by date_id)\nfeatures = [f\"feature_{i:02d}\" for i in range(79) ]\nsum_null_count_per_date_id = null_count_per_date_id.with_columns(\n    null_count=pl.sum_horizontal(features)\n).select(\n    \"date_id\", \"null_count\"\n).join( records_date_id, on=\"date_id\", how=\"inner\").to_pandas()\nsum_null_count_per_date_id[\"null_ratio\"] = sum_null_count_per_date_id[\"null_count\"] / sum_null_count_per_date_id[\"num_records\"] / 79\nsum_null_count_per_date_id","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T13:06:00.138176Z","iopub.execute_input":"2024-12-14T13:06:00.138566Z","iopub.status.idle":"2024-12-14T13:06:00.243408Z","shell.execute_reply.started":"2024-12-14T13:06:00.138534Z","shell.execute_reply":"2024-12-14T13:06:00.242282Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}