{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":84493,"databundleVersionId":9871156,"sourceType":"competition"}],"dockerImageVersionId":30786,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Import Libraries","metadata":{}},{"cell_type":"code","source":"import glob \nimport gc \nimport re\nimport os\nimport sys\nimport random\nimport numpy as np\nimport polars as pl\nimport pandas as pd\nimport lightgbm as lgb\nimport xgboost as xgb\nimport catboost as cat \nfrom tqdm import tqdm \nimport seaborn as sns\nfrom typing import Tuple\n    # cat_pred = models[2].predict(feat)\n\nimport kaggle_evaluation.jane_street_inference_server\n\n\npd.set_option(\"display.max_rows\", 2000)\npd.options.display.float_format = \"{:,.6f}\".format\n\nSEED = 42\ndef seed_everything(seed=SEED):\n    os.environ[\"PYTHONHASHSEED\"] = str(seed)\n    random.seed(seed)\n    np.random.seed(seed)\n\nseed_everything()","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-23T07:34:06.066698Z","iopub.execute_input":"2024-12-23T07:34:06.067005Z","iopub.status.idle":"2024-12-23T07:34:11.058226Z","shell.execute_reply.started":"2024-12-23T07:34:06.066966Z","shell.execute_reply":"2024-12-23T07:34:11.057538Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Global Settings","metadata":{}},{"cell_type":"markdown","source":"## Constants & Files","metadata":{}},{"cell_type":"code","source":"N_ESTIMATORS = 8_000\nN_SPLITS = 5\nEARLY_STOP = 100\n### DATA ###\n# Train data is too big to be loaded in the kaggle notebook. Make some adjustments if you want.\n# date_id > 1100\nNUM_ROWS_NOT_TO_BE_USED = 25023058 if N_ESTIMATORS > 10 else 39023058\nNUM_VALID_DATES = 100 if N_ESTIMATORS > 10 else 10\n\nINPUT_DIR = '/kaggle/input/jane-street-real-time-market-data-forecasting'\nOUTPUT_DIR = '/kaggle/output'","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T07:34:11.060456Z","iopub.execute_input":"2024-12-23T07:34:11.061319Z","iopub.status.idle":"2024-12-23T07:34:11.065542Z","shell.execute_reply.started":"2024-12-23T07:34:11.061278Z","shell.execute_reply":"2024-12-23T07:34:11.064648Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Columns(Features)","metadata":{}},{"cell_type":"code","source":"TARGET = 'responder_6'\nTIME_COLS = ['date_id', 'time_id']\nLEAD_COLS = ['symbol_id', 'weight']\nRESPONDER_COLS = [f\"responder_{i}\" for i in range(9)]\nFEAT_COLS = [f\"feature_{i:02d}\" for i in range(79)]\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T07:34:11.066619Z","iopub.execute_input":"2024-12-23T07:34:11.067375Z","iopub.status.idle":"2024-12-23T07:34:11.077872Z","shell.execute_reply.started":"2024-12-23T07:34:11.067347Z","shell.execute_reply":"2024-12-23T07:34:11.077168Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Custom eval metric functions\n- Made some changes to [yuanzhe zhou](https://www.kaggle.com/code/yuanzhezhou/jane-street-baseline-lgb-xgb-and-catboost)'s eval function","metadata":{}},{"cell_type":"code","source":"# Edited by kcy4\n# Removed r2_lgb and r2_xgb then merged them into one function.\n# Changed to use lgb.Dataset or xgb.DMatrix\ndef r2_gbt(y_pred, dtrain: lgb.Dataset|xgb.DMatrix):\n    y_true = dtrain.get_label()\n    weight = dtrain.get_weight()\n    r2 = 1 - np.average((y_pred - y_true) ** 2, weights=weight) / (np.average((y_true) ** 2, weights=weight) + 1e-38)\n    if isinstance(dtrain, lgb.Dataset):\n        return 'r2', r2, True\n    else: # for xgboost\n        return 'r2', -r2\n\n# No touch\n# Custom R2 metric for CatBoost\nclass r2_cbt(object):\n    def get_final_error(self, error, weight):\n        return 1 - error / (weight + 1e-38)\n\n    def is_max_optimal(self):\n        return True\n\n    def evaluate(self, approxes, target, weight):\n        assert len(approxes) == 1\n        assert len(target) == len(approxes[0])\n\n        approx = approxes[0]\n\n        error_sum = 0.0\n        weight_sum = 0.0\n\n        for i in range(len(approx)):\n            w = 1.0 if weight is None else weight[i]\n            weight_sum += w * (target[i] ** 2)\n            error_sum += w * ((approx[i] - target[i]) ** 2)\n\n        return error_sum, weight_sum","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T07:34:11.078781Z","iopub.execute_input":"2024-12-23T07:34:11.079015Z","iopub.status.idle":"2024-12-23T07:34:11.089352Z","shell.execute_reply.started":"2024-12-23T07:34:11.078991Z","shell.execute_reply":"2024-12-23T07:34:11.088496Z"},"jupyter":{"source_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Model params","metadata":{}},{"cell_type":"code","source":"MODEL_NAMES = ['lgb', 'xgb', 'cat'][:1]\n\nLGB_PARAMS = {\n    'objective'            : 'l2',\n    'boosting_type'        : 'gbdt',\n    'learning_rate'        : 0.02,\n    'num_leaves'           : 63,\n    'verbose'              : -1,\n    'random_state'         : SEED,\n    # 'device'               : 'gpu',\n    # 'gpu_platform_id'      : 0,\n    # 'gpu_device_id'        : 0,\n}\n\nXGB_PARAMS = {\n    'objective'            : 'reg:squarederror',\n    'eval_metric'          : 'rmse',\n    'disable_default_eval_metric': True,\n    # 'device'               : 'cuda:0',\n    'tree_method'          : 'hist',\n    'learning_rate'        : 0.05,\n    # 'eval_metric'          : r2_gbt,\n}\n\nCAT_PARAMS={\n    # 'task_type'            : 'GPU',\n    'loss_function'        : 'RMSE',\n    'eval_metric'          : r2_cbt(),\n    'n_estimators'         : N_ESTIMATORS,\n    'learning_rate'        : 0.05,\n    'verbose'              : 0,\n    'random_state'         : SEED,\n    'early_stopping_rounds': 100,\n}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T07:34:11.090439Z","iopub.execute_input":"2024-12-23T07:34:11.090730Z","iopub.status.idle":"2024-12-23T07:34:11.102100Z","shell.execute_reply.started":"2024-12-23T07:34:11.090705Z","shell.execute_reply":"2024-12-23T07:34:11.101273Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# time_df = pd.read_parquet(f\"{INPUT_DIR}/train.parquet\", columns=TIME_COLS)[NUM_ROWS_NOT_TO_BE_USED:].astype(np.uint16)\n# time_df[time_df['date_id'] <= 1100].shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T07:34:11.103087Z","iopub.execute_input":"2024-12-23T07:34:11.103422Z","iopub.status.idle":"2024-12-23T07:34:11.113611Z","shell.execute_reply.started":"2024-12-23T07:34:11.103383Z","shell.execute_reply":"2024-12-23T07:34:11.112801Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Data","metadata":{}},{"cell_type":"code","source":"def get_df():\n    print('Time data loading...')\n    time_df = pd.read_parquet(f\"{INPUT_DIR}/train.parquet\", columns=TIME_COLS)[NUM_ROWS_NOT_TO_BE_USED:].astype(np.uint16)\n    print('Lead data loading...')\n    lead_df = pd.read_parquet(f\"{INPUT_DIR}/train.parquet\", columns=LEAD_COLS)[NUM_ROWS_NOT_TO_BE_USED:]\n    lead_df['symbol_id'] = lead_df['symbol_id'].astype(np.uint32)\n    lead_df['weight'] = lead_df['weight'].astype(np.float32)\n    print('Target data loading...')\n    responder_6_df = pd.read_parquet(f\"{INPUT_DIR}/train.parquet\", columns=[TARGET])[NUM_ROWS_NOT_TO_BE_USED:].astype(np.float32)\n    print('Feature data loading...')\n    feat_dfs = []\n    num_chunk = 10\n    chunk_unit_len = (len(FEAT_COLS) // 10) + 1\n    for i in tqdm(range(10), total=10):\n        read_feat_cols = FEAT_COLS[i*chunk_unit_len:(i+1)*chunk_unit_len]\n        feat_dfs.append(pd.read_parquet(f\"{INPUT_DIR}/train.parquet\", columns=read_feat_cols)[NUM_ROWS_NOT_TO_BE_USED:].astype(np.float32))\n\n    return pd.concat([time_df]+[lead_df]+feat_dfs +[responder_6_df], axis=1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T07:34:11.116535Z","iopub.execute_input":"2024-12-23T07:34:11.117100Z","iopub.status.idle":"2024-12-23T07:34:11.126771Z","shell.execute_reply.started":"2024-12-23T07:34:11.117073Z","shell.execute_reply":"2024-12-23T07:34:11.125951Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Train & Infer function for lightgbm, xgboost, catboost","metadata":{}},{"cell_type":"code","source":"def train_model(df, model_names):\n    models = [None, None, None]\n    dates = df['date_id'].unique()\n    train_dates = dates[:-NUM_VALID_DATES]\n    valid_dates = dates[-NUM_VALID_DATES:]\n    \n    tr_weight = df['weight'].loc[df['date_id'].isin(train_dates)]\n    val_weight = df['weight'].loc[df['date_id'].isin(valid_dates)]\n\n    tr_y = df.loc[df['date_id'].isin(train_dates), TARGET]\n    val_y = df.loc[df['date_id'].isin(valid_dates), TARGET]\n\n    tr_X = df.loc[df['date_id'].isin(train_dates)][FEAT_COLS]\n    val_X = df.loc[df['date_id'].isin(valid_dates)][FEAT_COLS]\n\n    if 'lgb' in model_names:\n        print('lgb model training started. It takes some minutes. Have a coffee!')\n        tr_ds = lgb.Dataset(tr_X, label=tr_y, weight=tr_weight)\n        te_ds = lgb.Dataset(val_X, label=val_y, weight=val_weight, reference=tr_ds)\n        model = lgb.train(\n            LGB_PARAMS, tr_ds, N_ESTIMATORS, valid_sets=[te_ds], feval=r2_gbt,\n            callbacks=[lgb.early_stopping(EARLY_STOP)]\n        )\n        print(f\"lgb: {model.best_score['valid_0']['r2']:.06f}\")\n        del tr_ds, te_ds\n        gc.collect()\n        models[0] = model\n    if 'xgb' in model_names:\n        print('xgb model training started. It takes some minutes. Have a coffee!')\n        tr_ds = xgb.DMatrix(tr_X, label=tr_y, weight=tr_weight)\n        te_ds = xgb.DMatrix(val_X, label=val_y, weight=val_weight)\n        model = xgb.train(XGB_PARAMS, tr_ds, N_ESTIMATORS, evals=[(te_ds, 'eval')], early_stopping_rounds=EARLY_STOP, verbose_eval=False, custom_metric=r2_gbt)\n        print(f\"xgb: {-model.best_score:.06f}\")\n        del tr_ds, te_ds\n        gc.collect()\n        models[1] = model \n    if 'cat' in model_names:\n        print('cat model training started. It takes some minutes. Have a coffee!')\n        tr_ds = cat.Pool(tr_X, label=tr_y, weight=tr_weight)\n        te_ds = cat.Pool(val_X, label=val_y, weight=val_weight)\n        model = cat.CatBoostRegressor(**CAT_PARAMS)\n        model.fit(tr_ds, eval_set=te_ds, verbose=False)\n        print(f\"cat: {model.best_score_['validation']['r2_cbt']:.06f}\")\n        del tr_ds, te_ds\n        gc.collect()\n        models[2] = model\n        \n    return models\n\ndef infer(data, models):\n    return np.mean([model.predict(data) for model in models], axis=0)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T07:34:11.127585Z","iopub.execute_input":"2024-12-23T07:34:11.127785Z","iopub.status.idle":"2024-12-23T07:34:11.138944Z","shell.execute_reply.started":"2024-12-23T07:34:11.127763Z","shell.execute_reply":"2024-12-23T07:34:11.138161Z"},"jupyter":{"source_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Load Data & Train Models","metadata":{}},{"cell_type":"code","source":"# df = get_df()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T07:34:11.139955Z","iopub.execute_input":"2024-12-23T07:34:11.140222Z","iopub.status.idle":"2024-12-23T07:34:11.150425Z","shell.execute_reply.started":"2024-12-23T07:34:11.140198Z","shell.execute_reply":"2024-12-23T07:34:11.149608Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# models = train_model(df, MODEL_NAMES)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T07:34:11.151279Z","iopub.execute_input":"2024-12-23T07:34:11.151599Z","iopub.status.idle":"2024-12-23T07:34:11.160604Z","shell.execute_reply.started":"2024-12-23T07:34:11.151561Z","shell.execute_reply":"2024-12-23T07:34:11.159926Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Submit","metadata":{}},{"cell_type":"code","source":"count = 0\nmodels: list = None\nlags_ : pl.DataFrame | None = None","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T07:34:11.161735Z","iopub.execute_input":"2024-12-23T07:34:11.161998Z","iopub.status.idle":"2024-12-23T07:34:11.170037Z","shell.execute_reply.started":"2024-12-23T07:34:11.161973Z","shell.execute_reply":"2024-12-23T07:34:11.169057Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Replace this function with your inference code.\n# You can return either a Pandas or Polars dataframe, though Polars is recommended.\n# Each batch of predictions (except the very first) must be returned within 10 minutes of the batch features being provided.\ndef predict(test: pl.DataFrame, lags: pl.DataFrame | None) -> pl.DataFrame | pd.DataFrame:\n    \"\"\"Make a prediction.\"\"\"\n    # All the responders from the previous day are passed in at time_id == 0. We save them in a global variable for access at every time_id.\n    # Use them as extra features, if you like.\n    global lags_, models, count \n    if lags is not None:\n        lags_ = lags\n\n    if count == 0:\n        print('[1] Loading data...')\n        df = get_df()\n        print('data-shape:', df.shape)\n        print('[2] Training started')\n        models = train_model(df, MODEL_NAMES)\n        del df\n        gc.collect()\n\n    count += 1\n\n    predictions = test.select('row_id',pl.lit(0.0).alias('responder_6'))\n    \n    feat = test[FEAT_COLS].to_pandas()\n    lgb_pred = models[0].predict(feat)\n    # xgb_pred = models[1].predict(xgb.DMatrix(feat))\n    # cat_pred = models[2].predict(feat)\n    # pred = [lgb_pred, xgb_pred, cat_pred]\n    # pred = np.mean(pred, axis=0)\n    pred = lgb_pred\n    \n    predictions = predictions.with_columns(pl.Series('responder_6', pred))\n\n    # The predict function must return a DataFrame\n    assert isinstance(predictions, pl.DataFrame | pd.DataFrame)\n    # with columns 'row_id', 'responer_6'\n    assert predictions.columns == ['row_id', 'responder_6']\n    # and as many rows as the test data.\n    assert len(predictions) == len(test)\n\n    return predictions","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T07:34:11.171031Z","iopub.execute_input":"2024-12-23T07:34:11.171341Z","iopub.status.idle":"2024-12-23T07:34:11.181477Z","shell.execute_reply.started":"2024-12-23T07:34:11.171317Z","shell.execute_reply":"2024-12-23T07:34:11.180651Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"inference_server = kaggle_evaluation.jane_street_inference_server.JSInferenceServer(predict)\n\nif os.getenv('KAGGLE_IS_COMPETITION_RERUN'):\n    inference_server.serve()\nelse:\n    inference_server.run_local_gateway(\n        (\n            '/kaggle/input/jane-street-real-time-market-data-forecasting/test.parquet',\n            '/kaggle/input/jane-street-real-time-market-data-forecasting/lags.parquet',\n        )\n    )","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T07:34:11.182564Z","iopub.execute_input":"2024-12-23T07:34:11.183130Z","iopub.status.idle":"2024-12-23T08:05:20.197363Z","shell.execute_reply.started":"2024-12-23T07:34:11.183069Z","shell.execute_reply":"2024-12-23T08:05:20.196522Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}