{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceType":"competition","sourceId":84493,"databundleVersionId":11305158},{"sourceType":"datasetVersion","sourceId":10049844,"datasetId":5964697,"databundleVersionId":10323942},{"sourceType":"kernelVersion","sourceId":208423910}],"dockerImageVersionId":30786,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport polars as pl\nimport lightgbm as lgb","metadata":{"execution":{"iopub.status.busy":"2024-11-27T17:04:33.09785Z","iopub.execute_input":"2024-11-27T17:04:33.098286Z","iopub.status.idle":"2024-11-27T17:04:33.104094Z","shell.execute_reply.started":"2024-11-27T17:04:33.098237Z","shell.execute_reply":"2024-11-27T17:04:33.102986Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"#### Read data","metadata":{}},{"cell_type":"code","source":"# Read data\nroot_path = '/kaggle/input/jane-street-real-time-market-data-forecasting'\npl_df = pl.scan_parquet(f'{root_path}/train.parquet')\n\n# Select training columns\nall_cols = pl_df.collect_schema().names()\ntrain_cols = [col for col in all_cols if 'feature' in col]","metadata":{"execution":{"iopub.status.busy":"2024-11-27T17:01:30.282061Z","iopub.execute_input":"2024-11-27T17:01:30.282669Z","iopub.status.idle":"2024-11-27T17:01:30.418093Z","shell.execute_reply.started":"2024-11-27T17:01:30.282629Z","shell.execute_reply":"2024-11-27T17:01:30.416781Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"#### Custom functions","metadata":{}},{"cell_type":"code","source":"def weighted_r2_loss(y_true, y_pred, sample_weight):\n    \"\"\"\n    Loss function for sample-weighted zero-mean R2 score\n    \"\"\"\n    residuals = y_pred - y_true\n    weighted_residual_sum = np.sum(sample_weight * (residuals ** 2))\n\n    # Avoid division by zero\n    if weighted_residual_sum == 0:\n        weighted_residual_sum = 1e-10\n\n    grad = 2 * sample_weight * residuals / weighted_residual_sum\n    hess = 2 * sample_weight / weighted_residual_sum\n\n    return grad, hess\n    \ndef lgb_weighted_r2(y_true, y_pred, sample_weight):\n    \"\"\"\n    LGB specific evaluation function for sample-weighted zero-mean R2 score\n    \"\"\"\n    # Calculate weighted sum of squared residuals (numerator)\n    residuals = y_pred - y_true\n    weighted_residual_sum = np.sum(sample_weight * residuals ** 2)\n\n    # Calculate weighted sum of squared true values (denominator)\n    weighted_true_sum = np.sum(sample_weight * y_true ** 2)\n\n    # Calculate weighted R2\n    w_r2 = 1 - weighted_residual_sum / weighted_true_sum\n\n    is_higher_better = True\n\n    return 'weighted_r2', w_r2, is_higher_better","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T17:01:30.419437Z","iopub.execute_input":"2024-11-27T17:01:30.419807Z","iopub.status.idle":"2024-11-27T17:01:30.42744Z","shell.execute_reply.started":"2024-11-27T17:01:30.419773Z","shell.execute_reply":"2024-11-27T17:01:30.426251Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"#### Train model with custom loss and evaluation metric","metadata":{}},{"cell_type":"code","source":"# Create train / validation sets for demo purpose\nX_train = (pl_df\n        .filter(\n            pl.col('date_id').ge(1188)\n            & pl.col('date_id').le(1443)\n            )\n        .select(train_cols + ['weight', 'responder_6'])\n        .collect().to_pandas()\n    )\ny_train = X_train.pop('responder_6')\ntrain_weights = X_train.pop('weight')\n\nX_val = (pl_df\n    .filter(\n        pl.col('date_id').ge(1444)\n        & pl.col('date_id').le(1698)\n        )\n    .select(train_cols + ['weight', 'responder_6'])\n    .collect().to_pandas()\n)\ny_val = X_val.pop('responder_6')\nval_weights = X_val.pop('weight')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T17:01:30.429166Z","iopub.execute_input":"2024-11-27T17:01:30.429685Z","iopub.status.idle":"2024-11-27T17:02:01.673179Z","shell.execute_reply.started":"2024-11-27T17:01:30.429622Z","shell.execute_reply":"2024-11-27T17:02:01.671981Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"params = {\n        'objective': weighted_r2_loss,\n        'metric': 'None',\n        }\nmodel = lgb.LGBMRegressor(**params)\nmodel.fit(X_train, y_train, \n        eval_set=[(X_val, y_val)], \n        callbacks=[lgb.early_stopping(50)],\n        sample_weight=train_weights, \n        eval_sample_weight=[val_weights],\n        eval_metric=lgb_weighted_r2,\n        )","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T17:10:09.705948Z","iopub.execute_input":"2024-11-27T17:10:09.706799Z","iopub.status.idle":"2024-11-27T17:19:13.602838Z","shell.execute_reply.started":"2024-11-27T17:10:09.70676Z","shell.execute_reply":"2024-11-27T17:19:13.601733Z"}},"outputs":[],"execution_count":null}]}