{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84493,"databundleVersionId":9871156,"sourceType":"competition"}],"dockerImageVersionId":30786,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import polars as pl\nimport numpy as np\nimport pandas as pd\nfrom pathlib import Path\n\nfrom sklearn.linear_model import LinearRegression","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-11-18T22:10:56.571098Z","iopub.execute_input":"2024-11-18T22:10:56.571493Z","iopub.status.idle":"2024-11-18T22:10:56.577313Z","shell.execute_reply.started":"2024-11-18T22:10:56.571461Z","shell.execute_reply":"2024-11-18T22:10:56.576256Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"BASE_PATH = Path('/kaggle/input/jane-street-real-time-market-data-forecasting/train.parquet')\n\ntrain_ds = pl.read_parquet(BASE_PATH / 'partition_id=8' / 'part-0.parquet')\nval_ds = pl.read_parquet(BASE_PATH / 'partition_id=9' / 'part-0.parquet')\n\ntrain_ds.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-18T22:10:59.216066Z","iopub.execute_input":"2024-11-18T22:10:59.216535Z","iopub.status.idle":"2024-11-18T22:11:13.785280Z","shell.execute_reply.started":"2024-11-18T22:10:59.216495Z","shell.execute_reply":"2024-11-18T22:11:13.783919Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_train = train_ds.select([f'feature_{i:02d}' for i in range(79)]).fill_null(0).to_numpy()\nX_val = val_ds.select([f'feature_{i:02d}' for i in range(79)]).fill_null(0).to_numpy()\n\ny_train = train_ds['responder_6'].fill_null(0).to_numpy()\ny_val = val_ds['responder_6'].fill_null(0).to_numpy()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-18T22:15:16.773193Z","iopub.execute_input":"2024-11-18T22:15:16.774202Z","iopub.status.idle":"2024-11-18T22:15:16.859190Z","shell.execute_reply.started":"2024-11-18T22:15:16.774129Z","shell.execute_reply":"2024-11-18T22:15:16.857670Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model = LinearRegression().fit(X_train, y_train)\npred_val = model.predict(X_val).clip(-5, 5)\npred_val","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-18T22:13:47.920076Z","iopub.execute_input":"2024-11-18T22:13:47.920478Z","iopub.status.idle":"2024-11-18T22:14:24.362633Z","shell.execute_reply.started":"2024-11-18T22:13:47.920444Z","shell.execute_reply":"2024-11-18T22:14:24.358982Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.metrics import r2_score\n\nr2_score(y_val, pred_val, sample_weight=val_ds['weight'].to_numpy())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-18T22:15:19.724781Z","iopub.execute_input":"2024-11-18T22:15:19.725731Z","iopub.status.idle":"2024-11-18T22:15:19.850413Z","shell.execute_reply.started":"2024-11-18T22:15:19.725687Z","shell.execute_reply":"2024-11-18T22:15:19.848939Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import gc\n\ndel train_ds, val_ds, X_train, y_train, X_val, y_val\ngc.collect()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-18T22:16:44.032568Z","iopub.execute_input":"2024-11-18T22:16:44.032964Z","iopub.status.idle":"2024-11-18T22:16:44.581509Z","shell.execute_reply.started":"2024-11-18T22:16:44.032920Z","shell.execute_reply":"2024-11-18T22:16:44.574050Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"val_ds = None\nr2_scores = []\nfor i in range(9):\n    if val_ds is None:\n        train_ds = pl.read_parquet(BASE_PATH / f'partition_id={i}' / 'part-0.parquet')\n    else:\n        train_ds = val_ds\n    val_ds = pl.read_parquet(BASE_PATH / f'partition_id={i+1}' / 'part-0.parquet')\n    \n    X_train = train_ds.select([f'feature_{i:02d}' for i in range(79)]).fill_null(0).to_numpy()\n    X_val = val_ds.select([f'feature_{i:02d}' for i in range(79)]).fill_null(0).to_numpy()\n    \n    y_train = train_ds['responder_6'].fill_null(0).to_numpy()\n    y_val = val_ds['responder_6'].fill_null(0).to_numpy()\n\n    model = LinearRegression().fit(X_train, y_train)\n    pred_val = model.predict(X_val).clip(-5, 5)\n    r2_scores.append(r2_score(y_val, pred_val, sample_weight=val_ds['weight'].to_numpy()))\n\n    del train_ds, X_train, y_train, X_val, y_val\n    gc.collect()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-18T22:22:54.855247Z","iopub.execute_input":"2024-11-18T22:22:54.855613Z","iopub.status.idle":"2024-11-18T22:27:20.751957Z","shell.execute_reply.started":"2024-11-18T22:22:54.855583Z","shell.execute_reply":"2024-11-18T22:27:20.750635Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"r2_scores","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-18T22:27:58.562927Z","iopub.execute_input":"2024-11-18T22:27:58.563807Z","iopub.status.idle":"2024-11-18T22:27:58.571056Z","shell.execute_reply.started":"2024-11-18T22:27:58.563759Z","shell.execute_reply":"2024-11-18T22:27:58.569762Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"np.mean(r2_scores)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-18T22:27:40.433263Z","iopub.execute_input":"2024-11-18T22:27:40.433672Z","iopub.status.idle":"2024-11-18T22:27:40.441772Z","shell.execute_reply.started":"2024-11-18T22:27:40.433628Z","shell.execute_reply":"2024-11-18T22:27:40.440473Z"}},"outputs":[],"execution_count":null}]}