{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"codemirror_mode":{"name":"ipython","version":3},"file_extension":".py","mimetype":"text/x-python","name":"python","nbconvert_exporter":"python","pygments_lexer":"ipython3","version":"3.10.14"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":84493,"databundleVersionId":9871156,"sourceType":"competition"},{"sourceId":216349849,"sourceType":"kernelVersion"}],"dockerImageVersionId":30787,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true},"papermill":{"default_parameters":{},"duration":7.594014,"end_time":"2024-10-10T11:58:36.355301","environment_variables":{},"exception":null,"input_path":"__notebook__.ipynb","output_path":"__notebook__.ipynb","parameters":{},"start_time":"2024-10-10T11:58:28.761287","version":"2.6.0"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"**Note**  \n**Training Notebook cannot run at kaggle platform (because many memory is requied to run).**  \n**If you want to execute this code, you need to prepare own computations (out of kaggle).**  \n\n# Baseline notebooks:\n- Preprocessing : https://www.kaggle.com/code/motono0223/js24-preprocessing-create-lags\n- Training (Code only) : **this notebook** https://www.kaggle.com/code/motono0223/js24-train-gbdt-model-with-lags-singlemodel\n  - trained model : https://www.kaggle.com/datasets/motono0223/js24-trained-gbdt-model\n- Inference : https://www.kaggle.com/code/motono0223/js24-inference-gbdt-with-lags-singlemodel\n- EDA(1) : https://www.kaggle.com/code/motono0223/eda-jane-street-real-time-market-data-forecasting\n- EDA(2) : https://www.kaggle.com/code/motono0223/eda-v2-jane-street-real-time-market-forecasting","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport polars as pl\nimport numpy as np\nimport os\nfrom tqdm.auto import tqdm\nfrom matplotlib import pyplot as plt\nimport pickle\n\nfrom sklearn.metrics import r2_score\nfrom lightgbm import LGBMRegressor\nimport lightgbm as lgb\nfrom xgboost import XGBRegressor\nfrom catboost import CatBoostRegressor\nfrom sklearn.ensemble import VotingRegressor\n\nimport warnings\nwarnings.filterwarnings('ignore')\npd.options.display.max_columns = None\n\nimport kaggle_evaluation.jane_street_inference_server","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Configurations","metadata":{}},{"cell_type":"code","source":"class CONFIG:\n    seed = 42\n    target_col = \"responder_6\"\n    feature_cols = [\"symbol_id\", \"time_id\"] \\\n        + [f\"feature_{idx:02d}\" for idx in range(79)] \\\n        + [f\"responder_{idx}_lag_1\" for idx in range(9)]\n    categorical_cols = []","metadata":{"execution":{"iopub.status.busy":"2025-01-09T11:39:38.709854Z","iopub.execute_input":"2025-01-09T11:39:38.710315Z","iopub.status.idle":"2025-01-09T11:39:38.716315Z","shell.execute_reply.started":"2025-01-09T11:39:38.710275Z","shell.execute_reply":"2025-01-09T11:39:38.715047Z"},"trusted":true},"outputs":[],"execution_count":3},{"cell_type":"markdown","source":"# Load Data","metadata":{}},{"cell_type":"code","source":"train = pl.scan_parquet(\"/kaggle/input/js24-preprocessing-create-lags/training.parquet\").collect().to_pandas()\nvalid = pl.scan_parquet(\"/kaggle/input/js24-preprocessing-create-lags/validation.parquet\").collect().to_pandas()\n\ntrain.shape, valid.shape","metadata":{"execution":{"iopub.status.busy":"2025-01-09T11:39:42.401191Z","iopub.execute_input":"2025-01-09T11:39:42.401596Z","iopub.status.idle":"2025-01-09T11:40:26.849939Z","shell.execute_reply.started":"2025-01-09T11:39:42.40156Z","shell.execute_reply":"2025-01-09T11:40:26.848953Z"},"trusted":true},"outputs":[{"execution_count":4,"output_type":"execute_result","data":{"text/plain":"((21022056, 104), (1082224, 104))"},"metadata":{}}],"execution_count":4},{"cell_type":"code","source":"# Trick of boosting LB score: 0.45->0.49\ntrain = pd.concat([train, valid]).reset_index(drop=True)\ntrain.shape","metadata":{"trusted":true,"execution":{"execution_failed":"2025-01-09T11:40:57.251Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# GBDT models","metadata":{}},{"cell_type":"code","source":"def get_model(seed):\n    # XGBoost parameters\n    XGB_Params = {\n        'learning_rate': 0.05,\n        'max_depth': 6,\n        'n_estimators': 200,\n        'subsample': 0.8,\n        'colsample_bytree': 0.8,\n        'reg_alpha': 1,\n        'reg_lambda': 5,\n        'random_state': seed,\n        'tree_method': 'gpu_hist',\n        'device' : 'cuda',\n        'n_gpus' : 2,\n    }\n    \n    XGB_Model = XGBRegressor(**XGB_Params)\n    return XGB_Model","metadata":{"execution":{"execution_failed":"2025-01-09T11:40:57.252Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Training model","metadata":{}},{"cell_type":"code","source":"X_train = train[ CONFIG.feature_cols ]\ny_train = train[ CONFIG.target_col ]\nw_train = train[ \"weight\" ]\nX_valid = valid[ CONFIG.feature_cols ]\ny_valid = valid[ CONFIG.target_col ]\nw_valid = valid[ \"weight\" ]\n\nX_train.shape, y_train.shape, w_train.shape, X_valid.shape, y_valid.shape, w_valid.shape","metadata":{"execution":{"execution_failed":"2025-01-09T11:40:57.252Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time\nmodel = get_model(CONFIG.seed)\nmodel.fit( X_train, y_train, sample_weight=w_train)","metadata":{"execution":{"iopub.status.busy":"2024-10-28T10:56:36.176427Z","iopub.execute_input":"2024-10-28T10:56:36.176729Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y_pred_train1 = model.predict(X_train.iloc[:X_train.shape[0]//2])\ny_pred_train2 = model.predict(X_train.iloc[X_train.shape[0]//2:])\ntrain_score = r2_score(y_train, np.concatenate([y_pred_train1, y_pred_train2], axis=0), sample_weight=w_train )\ntrain_score","metadata":{"trusted":true,"execution":{"execution_failed":"2025-01-09T11:40:57.252Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y_pred_valid = model.predict(X_valid)\nvalid_score = r2_score(y_valid, y_pred_valid, sample_weight=w_valid )\nvalid_score","metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","papermill":{"duration":1.594977,"end_time":"2024-10-10T11:58:33.569648","exception":false,"start_time":"2024-10-10T11:58:31.974671","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y_means = { symbol_id : -1 for symbol_id in range(39) }\nfor symbol_id, gdf in train[[\"symbol_id\", CONFIG.target_col]].groupby(\"symbol_id\"):\n    y_mean = gdf[ CONFIG.target_col ].mean()\n    y_means[symbol_id] = y_mean\n    print(f\"symbol_id = {symbol_id}, y_means = {y_mean:.5f}\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"cv_detail = { symbol_id : 0 for symbol_id in range(39) }\nfor symbol_id, gdf in valid.groupby(\"symbol_id\"):\n    X_valid = gdf[ CONFIG.feature_cols ]\n    y_valid = gdf[ CONFIG.target_col ]\n    w_valid = gdf[ \"weight\" ]\n    y_pred_valid = model.predict(X_valid)\n    score = r2_score(y_valid, y_pred_valid, sample_weight=w_valid )\n    cv_detail[symbol_id] = score\n    \n    print(f\"symbol_id = {symbol_id}, score = {score:.5f}\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sids = list(cv_detail.keys())\nplt.bar(sids, [cv_detail[sid] for sid in sids])\nplt.grid()\nplt.xlabel(\"symbol_id\")\nplt.ylabel(\"CV score\")\nplt.show()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Save result","metadata":{}},{"cell_type":"code","source":"result = {\n    \"model\" : model,\n    \"cv\" : valid_score,\n    \"cv_detail\" : cv_detail,\n    \"y_mean\" : y_means,\n}\nwith open(\"result.pkl\", \"wb\") as fp:\n    pickle.dump(result, fp)","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}