{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":84493,"databundleVersionId":9871156,"sourceType":"competition"},{"sourceId":9883535,"sourceType":"datasetVersion","datasetId":6068987}],"dockerImageVersionId":30787,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# !pip install -q optuna\n# !pip install -q dask[dataframe]","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-11-13T10:29:22.852078Z","iopub.execute_input":"2024-11-13T10:29:22.852361Z","iopub.status.idle":"2024-11-13T10:29:22.856801Z","shell.execute_reply.started":"2024-11-13T10:29:22.852330Z","shell.execute_reply":"2024-11-13T10:29:22.855931Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport polars as pl\nimport pandas as pd","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-13T10:29:22.869503Z","iopub.execute_input":"2024-11-13T10:29:22.870018Z","iopub.status.idle":"2024-11-13T10:29:24.106002Z","shell.execute_reply.started":"2024-11-13T10:29:22.869985Z","shell.execute_reply":"2024-11-13T10:29:24.105172Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df = pl.read_parquet('/kaggle/input/js-2024-train/train_df_half.parquet')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-13T10:29:24.107513Z","iopub.execute_input":"2024-11-13T10:29:24.107911Z","iopub.status.idle":"2024-11-13T10:29:59.376462Z","shell.execute_reply.started":"2024-11-13T10:29:24.107878Z","shell.execute_reply":"2024-11-13T10:29:59.375659Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"features = ['symbol_id', 'weight'] + [col for col in df.columns if ('feature' in col) or ('_lag_' in col)]\nresponders = [col for col in df.columns if ('_lag_' in col)]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-13T10:29:59.377962Z","iopub.execute_input":"2024-11-13T10:29:59.378366Z","iopub.status.idle":"2024-11-13T10:29:59.384680Z","shell.execute_reply.started":"2024-11-13T10:29:59.378320Z","shell.execute_reply":"2024-11-13T10:29:59.383681Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Again taking 50% but we can comment it out if we got more RAM\ndf = df.slice(int(df.shape[0] * 0.5), None)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-13T10:29:59.386835Z","iopub.execute_input":"2024-11-13T10:29:59.387132Z","iopub.status.idle":"2024-11-13T10:29:59.406145Z","shell.execute_reply.started":"2024-11-13T10:29:59.387101Z","shell.execute_reply":"2024-11-13T10:29:59.405414Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X = df[features].slice(1, None)\n# X_r = df[responders].slice(1, None)\ny = df['responder_6'].slice(1, None)\nweights = df['weight'].slice(1, None)\ndf = None","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-13T10:29:59.407094Z","iopub.execute_input":"2024-11-13T10:29:59.407357Z","iopub.status.idle":"2024-11-13T10:29:59.428615Z","shell.execute_reply.started":"2024-11-13T10:29:59.407319Z","shell.execute_reply":"2024-11-13T10:29:59.427724Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import gc\ngc.collect()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-13T10:29:59.429613Z","iopub.execute_input":"2024-11-13T10:29:59.429870Z","iopub.status.idle":"2024-11-13T10:29:59.487033Z","shell.execute_reply.started":"2024-11-13T10:29:59.429841Z","shell.execute_reply":"2024-11-13T10:29:59.486021Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import lightgbm as lgb","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-13T10:29:59.488097Z","iopub.execute_input":"2024-11-13T10:29:59.488373Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"params = {'n_estimators': 680, 'learning_rate': 0.022167654221948718, 'max_depth': 5, 'num_leaves': 42, 'min_child_samples': 15, 'subsample': 0.8658662737679551, 'colsample_bytree': 0.6316624163538186, 'reg_alpha': 7.313049180654483, 'reg_lambda': 8.113493177057705}","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X.columns","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Create and train LightGBM model\nmodel = lgb.LGBMRegressor( **params, n_jobs=-1, verbose=-1, device = 'gpu')\nmodel.fit(X, y, sample_weight=weights)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def predict(test: pl.DataFrame, lags: pl.DataFrame | None) -> pd.DataFrame:\n    global lags_\n    \n    test_pd = test.to_pandas()\n    \n    if lags is not None:\n        lags_ = lags\n\n        # Convert lags to Pandas DataFrame if it's Polars\n        lags_pd = lags.to_pandas()\n        lags_pd = lags_pd.groupby(['date_id', 'symbol_id']).last().reset_index() \n        test_pd = test_pd.merge(lags_pd, on=['date_id', 'symbol_id'], how='left')\n    else:\n        \n        for idx in range(9):\n            test_pd[f'responder_{idx}_lag_1'] = 0.0\n\n    features = ['symbol_id', 'weight'] + [col for col in test_pd.columns if ('feature' in col) or ('responder' in col)]\n    print(features)  \n    # Prepare the features for prediction\n    predictions =  model.predict(test_pd[features])\n\n    # Create a DataFrame for the output\n    output = pd.DataFrame({\n        'row_id': test_pd['row_id'],\n        'responder_6': predictions\n    })\n\n    # Ensure the output DataFrame has the correct format\n    assert output.columns.tolist() == ['row_id', 'responder_6']\n    assert len(output) == len(test)\n\n    return output\n\nimport kaggle_evaluation.jane_street_inference_server\nimport os\n\n# Set up the inference server\ninference_server = kaggle_evaluation.jane_street_inference_server.JSInferenceServer(predict)\n\nif os.getenv('KAGGLE_IS_COMPETITION_RERUN'):\n    inference_server.serve()\nelse:\n    inference_server.run_local_gateway(\n        (\n            '/kaggle/input/jane-street-real-time-market-data-forecasting/test.parquet',\n            '/kaggle/input/jane-street-real-time-market-data-forecasting/lags.parquet',\n        )\n    )","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# import polars as pl\n# import kaggle_evaluation.jane_street_inference_server\n# import os\n\n# def predict(test: pl.DataFrame, lags: pl.DataFrame | None) -> pl.DataFrame:\n#     global lags_\n    \n#     if lags is not None:\n#         lags_ = lags\n#         # Group by and get last values using Polars\n#         lags_grouped = (lags\n#             .group_by(['date_id', 'symbol_id'])\n#             .agg(pl.all().last())\n#         )\n#         # Merge in Polars\n#         test = test.join(\n#             lags_grouped,\n#             on=['date_id', 'symbol_id'],\n#             how='left'\n#         )\n#     else:\n#         # Add zero-filled columns for responder lags\n#         for idx in range(9):\n#             test = test.with_columns(\n#                 pl.lit(0.0).alias(f'responder_{idx}_lag_1')\n#             )\n    \n#     # Get feature columns\n#     features = [col for col in test.columns if ('feature' in col) or ('responder' in col)]\n#     print(features)\n    \n#     # Convert just the features to numpy for model prediction\n#     predictions = model.predict(test.select(features).to_numpy()) ** 0.5\n    \n#     # Create output DataFrame directly in Polars\n#     output = pl.DataFrame({\n#         'row_id': test.get_column('row_id'),\n#         'responder_6': predictions\n#     })\n    \n#     # Assertions for validation\n#     assert output.columns == ['row_id', 'responder_6']\n#     assert len(output) == len(test)\n#     return output\n\n# # Set up the inference server\n# inference_server = kaggle_evaluation.jane_street_inference_server.JSInferenceServer(predict)\n\n# if os.getenv('KAGGLE_IS_COMPETITION_RERUN'):\n#     inference_server.serve()\n# else:\n#     inference_server.run_local_gateway(\n#         (\n#             '/kaggle/input/jane-street-real-time-market-data-forecasting/test.parquet',\n#             '/kaggle/input/jane-street-real-time-market-data-forecasting/lags.parquet',\n#         )\n#     )","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}