{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":84493,"databundleVersionId":11305158,"sourceType":"competition"},{"sourceId":203900450,"sourceType":"kernelVersion"}],"dockerImageVersionId":31041,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"### Train XGB\n\n* [1.CFG and Model Setup](#s1)\n\n* [2.Load Data](#s2)\n\n* [3.Fit](#s3)","metadata":{}},{"cell_type":"markdown","source":"#### Libs needed","metadata":{}},{"cell_type":"code","source":"import os\nfrom tqdm.auto import tqdm\nfrom matplotlib import pyplot as plt\nimport pickle\nimport joblib # dump model as .pkl\nimport pandas as pd\nimport polars as pl\n############################\nfrom sklearn.metrics import r2_score\n# from lightgbm import LGBMRegressor\n# import lightgbm as lgb\nfrom xgboost import XGBRegressor # This NB only focus on \n# from catboost import CatBoostRegressor\nfrom sklearn.ensemble import VotingRegressor\n##########################\nimport warnings\nwarnings.filterwarnings('ignore')\npd.options.display.max_columns = None\n#########################\n# import kaggle_evaluation.jane_street_inference_server","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-22T18:27:51.485904Z","iopub.execute_input":"2025-05-22T18:27:51.486142Z","iopub.status.idle":"2025-05-22T18:27:55.476351Z","shell.execute_reply.started":"2025-05-22T18:27:51.486116Z","shell.execute_reply":"2025-05-22T18:27:55.475542Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"#### 1. CFG and Model Setup <a class=\"anchor\"  id=\"s1\"></a>","metadata":{}},{"cell_type":"code","source":"# Configuration\n\nclass CFG: \n    # Note: this is convenient for \n    # updating data/versioning due to different input\n    # which is a very common use in Kaggle community\n    seed = 42\n    target_col = \"responder_6\"\n    feature_cols = [\"symbol_id\", \"time_id\"] \\\n        + [f\"feature_{idx:02d}\" for idx in range(79)] \\\n        + [f\"responder_{idx}_lag_1\" for idx in range(9)] # use the lag data\n    categorical_cols = []\n\n# Get model\n\ndef get_model(seed):\n    # XGBoost parameters\n    # easy to tune using this set-up\n    XGB_Params = {\n        'learning_rate': 0.05, # common use\n        'max_depth': 6, # <= sqrt(n_features)\n        'n_estimators': 200,\n        'subsample': 0.8,\n        'colsample_bytree': 0.8,\n        'reg_alpha': 1,\n        'reg_lambda': 5,\n        'random_state': seed,\n        'tree_method': 'gpu_hist',\n        'device' : 'cuda',\n        'n_gpus' : 2, # Must turn GPU t4 x 2 on !!!\n    }\n    \n    XGB_Model = XGBRegressor(**XGB_Params)\n    return XGB_Model","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-22T18:27:55.477296Z","iopub.execute_input":"2025-05-22T18:27:55.477699Z","iopub.status.idle":"2025-05-22T18:27:55.483689Z","shell.execute_reply.started":"2025-05-22T18:27:55.477678Z","shell.execute_reply":"2025-05-22T18:27:55.482732Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"#### 2. Load Data <a class=\"anchor\"  id=\"s2\"></a>","metadata":{}},{"cell_type":"code","source":"# Data loading\n\nDT_GT = 1500 \n# Note: Kaggle doesn't have enough RAM for full-size data\n# Full-size can be done using chunk-wise run in Colab Pro+\n\ntrain = pl.scan_parquet(\n    \"/kaggle/input/js24-preprocessing-create-lags/training.parquet\"\n).filter(pl.col(\"date_id\") > DT_GT).collect().to_pandas()\n\nvalid = pl.scan_parquet(\n    \"/kaggle/input/js24-preprocessing-create-lags/validation.parquet\"\n).filter(pl.col(\"date_id\") > DT_GT).collect().to_pandas()\n\n# train.shape, valid.shape\ntrain = pd.concat([train, valid]).reset_index(drop=True)\n\n# Train vs Valid\n\nX_train = train[ CFG.feature_cols ]\ny_train = train[ CFG.target_col ]\nw_train = train[ \"weight\" ]\nX_valid = valid[ CFG.feature_cols ]\ny_valid = valid[ CFG.target_col ]\nw_valid = valid[ \"weight\" ]\n\n(X_train.shape, y_train.shape, w_train.shape, X_valid.shape, y_valid.shape, w_valid.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-22T18:27:55.485647Z","iopub.execute_input":"2025-05-22T18:27:55.485903Z","iopub.status.idle":"2025-05-22T18:28:18.062822Z","shell.execute_reply.started":"2025-05-22T18:27:55.485883Z","shell.execute_reply":"2025-05-22T18:28:18.062181Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"#### 3. Fit <a class=\"anchor\"  id=\"s3\"></a>\n  - Note: We only provide one-fold train-validate considering time consumed and simliar performance to 5-fold  ","metadata":{}},{"cell_type":"code","source":"# Train\n# %%time\n\nTRAIN = False # Change to True if needed\n\nif TRAIN:\n\n    model = get_model(CFG.seed)\n    model.fit(X_train, y_train, sample_weight=w_train)\n    \n    # R2 Score\n    y_pred_train = model.predict(X_train)\n    train_score = r2_score(y_train, y_pred_train, sample_weight=w_train )\n    \n    y_pred_valid = model.predict(X_valid)\n    valid_score = r2_score(y_valid, y_pred_valid, sample_weight=w_valid )\n    \n    print(f\"Train R2: {train_score}, Validation R2: {valid_score}\")\n    \n    # Save\n    # joblib.dump(model, \"js24_xgb.pkl\") # download from the output after saving\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-22T18:35:51.223929Z","iopub.execute_input":"2025-05-22T18:35:51.224239Z","iopub.status.idle":"2025-05-22T18:35:51.228722Z","shell.execute_reply.started":"2025-05-22T18:35:51.224212Z","shell.execute_reply":"2025-05-22T18:35:51.228134Z"}},"outputs":[],"execution_count":null}]}