{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"codemirror_mode":{"name":"ipython","version":3},"file_extension":".py","mimetype":"text/x-python","name":"python","nbconvert_exporter":"python","pygments_lexer":"ipython3","version":"3.10.14"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84493,"databundleVersionId":11305158,"sourceType":"competition"}],"dockerImageVersionId":30786,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false},"papermill":{"default_parameters":{},"duration":4.669361,"end_time":"2024-10-10T13:05:46.686069","environment_variables":{},"exception":null,"input_path":"__notebook__.ipynb","output_path":"__notebook__.ipynb","parameters":{},"start_time":"2024-10-10T13:05:42.016708","version":"2.6.0"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport polars as pl\nimport numpy as np\nfrom scipy.stats import jarque_bera\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\nfrom sklearn.linear_model import LinearRegression\nfrom sklearn.metrics import mean_squared_error, r2_score\n\n# Evaluation\nimport os\nimport kaggle_evaluation.jane_street_inference_server as JS_eval","metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","papermill":{"duration":1.223703,"end_time":"2024-10-10T13:05:45.825911","exception":false,"start_time":"2024-10-10T13:05:44.602208","status":"completed"},"tags":[],"trusted":true,"execution":{"iopub.status.busy":"2025-03-10T20:55:09.314279Z","iopub.execute_input":"2025-03-10T20:55:09.315383Z","iopub.status.idle":"2025-03-10T20:55:11.750608Z","shell.execute_reply.started":"2025-03-10T20:55:09.315336Z","shell.execute_reply":"2025-03-10T20:55:11.749346Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Metadata Load\nfeatures_csv = pd.read_csv('/kaggle/input/jane-street-real-time-market-data-forecasting/features.csv') # metadata pertaining to the anonymized features\nresponders_csv = pd.read_csv('/kaggle/input/jane-street-real-time-market-data-forecasting/responders.csv') # metadata pertaining to the anonymized responders\nsample_submission_csv = pd.read_csv('/kaggle/input/jane-street-real-time-market-data-forecasting/sample_submission.csv') # format of the predictions your model should make (responder_6)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-10T20:55:13.836353Z","iopub.execute_input":"2025-03-10T20:55:13.836916Z","iopub.status.idle":"2025-03-10T20:55:13.874374Z","shell.execute_reply.started":"2025-03-10T20:55:13.836876Z","shell.execute_reply":"2025-03-10T20:55:13.873285Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# The submission must contain the prediction for each symbol_id, represented as row_id\nsample_submission_csv.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-10T20:55:15.330168Z","iopub.execute_input":"2025-03-10T20:55:15.330559Z","iopub.status.idle":"2025-03-10T20:55:15.355115Z","shell.execute_reply.started":"2025-03-10T20:55:15.330524Z","shell.execute_reply":"2025-03-10T20:55:15.354003Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Responder tag\nresponders_csv.loc[[6]] # just tag_2 == True","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-10T20:55:17.406350Z","iopub.execute_input":"2025-03-10T20:55:17.406732Z","iopub.status.idle":"2025-03-10T20:55:17.425380Z","shell.execute_reply.started":"2025-03-10T20:55:17.406700Z","shell.execute_reply":"2025-03-10T20:55:17.424218Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Extract objective features\ncols = ['feature', 'tag_2']\nfeatures_sel = features_csv[cols]\n\nfeatures_sel = features_sel.query('tag_2 == True')\nfeatures_obj = features_sel['feature'].unique()\nfeatures_obj","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-10T20:55:18.451295Z","iopub.execute_input":"2025-03-10T20:55:18.453198Z","iopub.status.idle":"2025-03-10T20:55:18.475899Z","shell.execute_reply.started":"2025-03-10T20:55:18.453135Z","shell.execute_reply":"2025-03-10T20:55:18.474066Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Objective columns\nts_columns = ['date_id', 'time_id', 'symbol_id', 'weight', 'responder_6'] + list(features_obj)\n\nX_columns = ['date_id', 'time_id', 'symbol_id', 'weight'] + list(features_obj)\ny_columns = ['date_id', 'time_id', 'symbol_id', 'responder_6']\n\nts_test_columns = ['row_id'] + X_columns\nts_lags_columns = ['date_id', 'time_id', 'symbol_id', 'responder_6_lag_1']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-10T20:55:54.620202Z","iopub.execute_input":"2025-03-10T20:55:54.620563Z","iopub.status.idle":"2025-03-10T20:55:54.626640Z","shell.execute_reply.started":"2025-03-10T20:55:54.620532Z","shell.execute_reply":"2025-03-10T20:55:54.625411Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Train: 9 partitions\ntrain_partitions = [\n    pl.read_parquet(f\"/kaggle/input/jane-street-real-time-market-data-forecasting/train.parquet/partition_id={p}/part-0.parquet\",\n                    columns=ts_columns)\n    for p in range(10)\n]\nts_train = pl.concat(train_partitions)\n\n# Test parquet\nX_ts_test = pl.read_parquet('/kaggle/input/jane-street-real-time-market-data-forecasting/test.parquet/date_id=0/part-0.parquet',\n                            columns=ts_test_columns)\n\n# Lags parquet\ny_ts_lags = pl.read_parquet('/kaggle/input/jane-street-real-time-market-data-forecasting/lags.parquet/date_id=0/part-0.parquet',\n                            columns=ts_lags_columns)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-10T21:24:07.385196Z","iopub.execute_input":"2025-03-10T21:24:07.386186Z","iopub.status.idle":"2025-03-10T21:24:11.549990Z","shell.execute_reply.started":"2025-03-10T21:24:07.386142Z","shell.execute_reply":"2025-03-10T21:24:11.548900Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Select the desired columns for X and y\nts_train = ts_train.drop_nulls()\n\nX = ts_train[X_columns]\ny = ts_train[y_columns]\n\nts_train","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-10T21:24:13.171174Z","iopub.execute_input":"2025-03-10T21:24:13.171619Z","iopub.status.idle":"2025-03-10T21:24:13.697936Z","shell.execute_reply.started":"2025-03-10T21:24:13.171583Z","shell.execute_reply":"2025-03-10T21:24:13.696676Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"del ts_train, features_csv, responders_csv, sample_submission_csv","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-10T21:24:15.149761Z","iopub.execute_input":"2025-03-10T21:24:15.150310Z","iopub.status.idle":"2025-03-10T21:24:15.157151Z","shell.execute_reply.started":"2025-03-10T21:24:15.150258Z","shell.execute_reply":"2025-03-10T21:24:15.155662Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(f\"Shape X:{np.shape(X)} \\nShape y:{np.shape(y)} \\n\\nNull X: {sum(X.null_count().sum().row(0))} \\nNull y: {sum(y.null_count().sum().row(0))}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-10T21:22:19.023064Z","iopub.execute_input":"2025-03-10T21:22:19.023760Z","iopub.status.idle":"2025-03-10T21:22:19.032212Z","shell.execute_reply.started":"2025-03-10T21:22:19.023715Z","shell.execute_reply":"2025-03-10T21:22:19.030897Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(f\"Number of symbol_id X: {len(X['symbol_id'].unique())} \\nNumber of symbol_id y: {len(y['symbol_id'].unique())}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-10T21:22:20.807963Z","iopub.execute_input":"2025-03-10T21:22:20.808359Z","iopub.status.idle":"2025-03-10T21:22:21.494399Z","shell.execute_reply.started":"2025-03-10T21:22:20.808327Z","shell.execute_reply":"2025-03-10T21:22:21.493095Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Arreglar outliers de salto repentino\n\nfor n in range(39):\n    df_plot = y.filter(pl.col('symbol_id') == n)\n    trajectory = np.cumsum(df_plot['responder_6'])\n    plt.plot(df_plot['date_id'], trajectory, linewidth=0.3)\n\nplt.xlim(0, y['date_id'].max())\nplt.xlabel('Date ID')\nplt.ylabel('Cumulative Sum')\nplt.title('Symbols Trajectories')\nplt.grid(True, linestyle=\"--\", linewidth=0.5, alpha=0.7)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-10T21:29:39.171862Z","iopub.execute_input":"2025-03-10T21:29:39.172238Z","iopub.status.idle":"2025-03-10T21:29:44.657038Z","shell.execute_reply.started":"2025-03-10T21:29:39.172205Z","shell.execute_reply":"2025-03-10T21:29:44.655878Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Jarque-Bera test for H0: normal distribution, H1: non-normal\njb = []\nfor n in range(39):\n    df_plot = y.filter(pl.col('symbol_id') == n)\n    trajectory = np.cumsum(df_plot['responder_6'])\n    jb.append(jarque_bera(trajectory)[1])\n\nmax(jb) # H1: Non-normal","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-10T21:28:29.542038Z","iopub.execute_input":"2025-03-10T21:28:29.542446Z","iopub.status.idle":"2025-03-10T21:28:31.330859Z","shell.execute_reply.started":"2025-03-10T21:28:29.542410Z","shell.execute_reply":"2025-03-10T21:28:31.329586Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Distribución predominante\nsns.histplot(trajectory, bins=50);","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-10T21:32:57.949774Z","iopub.execute_input":"2025-03-10T21:32:57.950200Z","iopub.status.idle":"2025-03-10T21:32:59.032122Z","shell.execute_reply.started":"2025-03-10T21:32:57.950165Z","shell.execute_reply":"2025-03-10T21:32:59.030846Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from scipy.stats import probplot\n\nprobplot(trajectory, dist=\"norm\", plot=plt)\nplt.gca().get_lines()[0].set_markersize(0.5);","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-10T21:39:18.000647Z","iopub.execute_input":"2025-03-10T21:39:18.001137Z","iopub.status.idle":"2025-03-10T21:39:19.594861Z","shell.execute_reply.started":"2025-03-10T21:39:18.001097Z","shell.execute_reply":"2025-03-10T21:39:19.593631Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Técnias a considerar: Lopez de Prado\n\nCapítulo 2: Financial Data Structures\n* Cumulative Sum Control Chart (CUSUM)\n* Supremum Augmented Dickey-Fuller (SADF)\n\nCapítulo 5: Fractional Differentiated Features\n* FFD(d), d entre[0,1]  (Más importante)\n\nCapítulo 7: Cross-Validation in Finance\n* Purged k-fold cv\n* Embargo  ","metadata":{}},{"cell_type":"code","source":"'''\n#Cummulative Sum (Diferenciación parcial previa necesaria)\nX_LR = X_LR.with_columns([\n    pl.col(c).cum_sum().alias(c) for c in X_LR.columns\n])\ny_LR = np.cumsum(y_LR)\n'''","metadata":{"execution":{"iopub.status.busy":"2025-03-08T22:12:30.129828Z","iopub.execute_input":"2025-03-08T22:12:30.130237Z","iopub.status.idle":"2025-03-08T22:12:30.137643Z","shell.execute_reply.started":"2025-03-08T22:12:30.130203Z","shell.execute_reply":"2025-03-08T22:12:30.136435Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Train Test with TS\n    # Filtrar y matchear index en y_\nX_LR = X.filter(pl.col('symbol_id') == 0)[:, 4:]\ny_LR = y.filter(pl.col('symbol_id') == 0)[:, 3]\n\ntrain_size = int(len(X_LR) * 0.8)\nX_train, X_test = X_LR[:train_size], X_LR[train_size:]\ny_train, y_test = y_LR[:train_size], y_LR[train_size:]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-10T21:39:51.228643Z","iopub.execute_input":"2025-03-10T21:39:51.229309Z","iopub.status.idle":"2025-03-10T21:39:51.387300Z","shell.execute_reply.started":"2025-03-10T21:39:51.229271Z","shell.execute_reply":"2025-03-10T21:39:51.385928Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Simplest model LR\nLM = LinearRegression()\nLM.fit(X_train, y_train)\n\ny_pred = LM.predict(X_test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-10T21:39:52.534950Z","iopub.execute_input":"2025-03-10T21:39:52.535355Z","iopub.status.idle":"2025-03-10T21:39:52.918442Z","shell.execute_reply.started":"2025-03-10T21:39:52.535320Z","shell.execute_reply":"2025-03-10T21:39:52.915839Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Resultados del modelo lineal (relaciones no lineales)\nprint(f\"MSE: {mean_squared_error(y_test, y_pred):.2f} \\nR²: {r2_score(y_test, y_pred):.2f}\")\n\n# Con datos en diferencias; MSE: 0.47 R^2: -0.00\n# Con diferenciación parcial: ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-10T21:39:54.520692Z","iopub.execute_input":"2025-03-10T21:39:54.521101Z","iopub.status.idle":"2025-03-10T21:39:54.532279Z","shell.execute_reply.started":"2025-03-10T21:39:54.521068Z","shell.execute_reply":"2025-03-10T21:39:54.531116Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"The evaluation API requires that you set up a server which will respond to inference requests. We have already defined the server; you just need write the predict function. When we evaluate your submission on the hidden test set the client defined in `jane_street_gateway` will run in a different container with direct access to the hidden test set and hand off the data timestep by timestep.\n\n\n\nYour code will always have access to the published copies of the files.","metadata":{"papermill":{"duration":0.002051,"end_time":"2024-10-10T13:05:45.83073","exception":false,"start_time":"2024-10-10T13:05:45.828679","status":"completed"},"tags":[]}},{"cell_type":"code","source":"'''\nlags_ : pl.DataFrame | None = None\n\n\n# Replace this function with your inference code.\n# You can return either a Pandas or Polars dataframe, though Polars is recommended.\n# Each batch of predictions (except the very first) must be returned within 1 minute of the batch features being provided.\ndef predict(test: pl.DataFrame, lags: pl.DataFrame | None) -> pl.DataFrame | pd.DataFrame:\n    \"\"\"Make a prediction.\"\"\"\n    # All the responders from the previous day are passed in at time_id == 0. We save them in a global variable for access at every time_id.\n    # Use them as extra features, if you like.\n    global lags_\n    if lags is not None:\n        lags_ = lags\n\n    # Replace this section with your own predictions\n    predictions = test.select(\n        'row_id',\n        pl.lit(0.0).alias('responder_6'),\n    )\n\n    if isinstance(predictions, pl.DataFrame):\n        assert predictions.columns == ['row_id', 'responder_6']\n    elif isinstance(predictions, pd.DataFrame):\n        assert (predictions.columns == ['row_id', 'responder_6']).all()\n    else:\n        raise TypeError('The predict function must return a DataFrame')\n    # Confirm has as many rows as the test data.\n    assert len(predictions) == len(test)\n\n    return predictions\n'''","metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","execution":{"iopub.status.busy":"2025-03-08T19:11:52.542082Z","iopub.execute_input":"2025-03-08T19:11:52.542416Z","iopub.status.idle":"2025-03-08T19:11:52.548230Z","shell.execute_reply.started":"2025-03-08T19:11:52.542384Z","shell.execute_reply":"2025-03-08T19:11:52.547499Z"},"papermill":{"duration":0.015917,"end_time":"2024-10-10T13:05:45.848958","exception":false,"start_time":"2024-10-10T13:05:45.833041","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"When your notebook is run on the hidden test set, inference_server.serve must be called within 15 minutes of the notebook starting or the gateway will throw an error. If you need more than 15 minutes to load your model you can do so during the very first `predict` call, which does not have the usual 1 minute response deadline.","metadata":{"papermill":{"duration":0.00196,"end_time":"2024-10-10T13:05:45.853279","exception":false,"start_time":"2024-10-10T13:05:45.851319","status":"completed"},"tags":[]}},{"cell_type":"code","source":"'''\ninference_server = kaggle_evaluation.jane_street_inference_server.JSInferenceServer(predict)\n\nif os.getenv('KAGGLE_IS_COMPETITION_RERUN'):\n    inference_server.serve()\nelse:\n    inference_server.run_local_gateway(\n        (\n            '/kaggle/input/jane-street-real-time-market-data-forecasting/test.parquet',\n            '/kaggle/input/jane-street-real-time-market-data-forecasting/lags.parquet',\n        )\n    )\n'''","metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","execution":{"iopub.status.busy":"2025-03-08T19:11:56.149723Z","iopub.execute_input":"2025-03-08T19:11:56.150031Z","iopub.status.idle":"2025-03-08T19:11:56.155454Z","shell.execute_reply.started":"2025-03-08T19:11:56.150005Z","shell.execute_reply":"2025-03-08T19:11:56.154530Z"},"papermill":{"duration":0.308219,"end_time":"2024-10-10T13:05:46.163573","exception":false,"start_time":"2024-10-10T13:05:45.855354","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":" ","metadata":{}},{"cell_type":"markdown","source":"### Citation\n\nMaanit Desai, Yirun Zhang, Ryan Holbrook, Kait O'Neil, and Maggie Demkin. Jane Street Real-Time Market Data Forecasting. https://kaggle.com/competitions/jane-street-real-time-market-data-forecasting, 2024. Kaggle.","metadata":{}}]}