{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84493,"databundleVersionId":9871156,"sourceType":"competition"},{"sourceId":10192259,"sourceType":"datasetVersion","datasetId":6297429}],"dockerImageVersionId":30804,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## On submission format:\nhttps://www.kaggle.com/code/gkitchen/jane-street-first-steps\n\n## based on JS_notebook_4","metadata":{}},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-27T17:42:36.898639Z","iopub.execute_input":"2024-12-27T17:42:36.899050Z","iopub.status.idle":"2024-12-27T17:42:37.320565Z","shell.execute_reply.started":"2024-12-27T17:42:36.899014Z","shell.execute_reply":"2024-12-27T17:42:37.319388Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport pandas as pd\nimport polars as pl\nimport pickle\nimport time\nfrom tqdm.auto import tqdm\n\nimport xgboost as xgb\n\nimport kaggle_evaluation.jane_street_inference_server# as js_server","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-27T17:42:37.322328Z","iopub.execute_input":"2024-12-27T17:42:37.322826Z","iopub.status.idle":"2024-12-27T17:42:38.155097Z","shell.execute_reply.started":"2024-12-27T17:42:37.322770Z","shell.execute_reply":"2024-12-27T17:42:38.154129Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model_path = \"/kaggle/input/trained-model-0/trained_model_0.pkl\"\nwith open(model_path, \"rb\") as fp:\n    model = pickle.load(fp)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-27T17:42:38.156267Z","iopub.execute_input":"2024-12-27T17:42:38.156785Z","iopub.status.idle":"2024-12-27T17:42:38.169583Z","shell.execute_reply.started":"2024-12-27T17:42:38.156732Z","shell.execute_reply":"2024-12-27T17:42:38.168706Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# variables to be used in predict()\nlags_ = None\ndays_passed = 0\nhistory_cache = []\ntime_step_count = 0\nscoring = False\nretrained = False\n\nfeature_cols = [\"symbol_id\", \"time_id\"] \\\n    + [f\"feature_{idx:02d}\" for idx in range(79)] \\\n    + [f\"responder_{idx}_lag_1\" for idx in range(9)]\n\ncols_needed_to_cache = ['date_id', 'weight'] + feature_cols\n\ntotal_time_steps = 1\npbar_length = total_time_steps\n\n# Model parameters (can be adjusted or expanded for tuning)\nseed = 42\nXGB_Params = {\n    'objective': 'reg:squarederror',\n    'learning_rate': 0.05,\n    'max_depth': 6,\n    'n_estimators': 200,\n    'subsample': 0.8,\n    'colsample_bytree': 0.8,\n    'reg_alpha': 1,\n    'reg_lambda': 5,\n    'random_state': seed,\n    'tree_method': 'hist',\n    'device': 'cuda',\n    'eval_metric': 'rmse',\n    #'early_stopping_rounds': 10,\n    'early_stopping_rounds': 5,\n}\n\n#placerholders for X, y, w\nX = np.zeros((0,90), dtype=np.float32)\ny = np.zeros((0,), dtype=np.float32)\nw = np.zeros((0,), dtype=np.float32)\n\npbar = tqdm(total=pbar_length, disable=(pbar_length == 0))\n#pbar.clear()    ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-27T17:42:38.171476Z","iopub.execute_input":"2024-12-27T17:42:38.172243Z","iopub.status.idle":"2024-12-27T17:42:38.195233Z","shell.execute_reply.started":"2024-12-27T17:42:38.172207Z","shell.execute_reply":"2024-12-27T17:42:38.194200Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def predict(test: pl.DataFrame, lags: pl.DataFrame | None) -> pl.DataFrame | pd.DataFrame:\n    \"\"\"Make a prediction.\"\"\"\n\n    global lags_, days_passed, history_cache, model, time_step_count, scoring, retrained, X, y, w\n\n    start = time.time()\n    \n    #print(test)\n\n    if (lags is not None) and (lags_ is not None):\n        #print('Both lag and lags_ not None')\n        lags_for_merge = lags_.clone()\n        #print(f'lags_for_merge:{lags_for_merge}')\n\n        if scoring:\n            #print('There is scoring')\n            days_passed += 1\n            print(f'days_passed: {days_passed}')\n\n            lags_for_merge.columns = [i + \"_fitting\" if \"res\" in i else i for i in lags_for_merge.columns]\n\n            lags_for_merge = lags_for_merge.with_columns([\n                    (pl.col(\"date_id\") - 1).alias(\"date_id\")\n                ])\n\n            #print(f'test[\"date_id\"]: {test[\"date_id\"]}')\n            if test['date_id'].to_numpy()[0] > 0:\n                print(f'test date_id > 0: {test[\"date_id\"].to_numpy()[0]}')\n                \n                test_last_date = pl.concat(history_cache)\n\n                history_cache = []\n\n                test_last_date_merged = test_last_date.join(lags_for_merge.select(['date_id', 'time_id', 'symbol_id', 'responder_6_lag_1_fitting']), how='inner', on=['date_id', 'time_id', 'symbol_id'])\n\n                X_one = np.array(test_last_date_merged.select(feature_cols).to_numpy())\n    \n                n_to_add = len(X_one)\n                        \n                X = np.vstack([X[n_to_add:], X_one])\n                \n                del X_one\n\n                w_one = test_last_date_merged.select(\"weight\")['weight'].to_numpy()\n                y_one = test_last_date_merged.select('responder_6_lag_1_fitting')['responder_6_lag_1_fitting'].to_numpy() \n\n                w = np.concatenate([w[n_to_add:], w_one])\n                y = np.concatenate([y[n_to_add:], y_one])\n\n    \n    if (lags is not None) and (lags_ is None):\n        print('There is lags but no lags_')\n        lags_ = lags\n        #print(f'lags_: {lags_}')\n        \n        #if lags_ is not None:\n        lags_for_features = lags_.group_by([\"date_id\", \"symbol_id\"], maintain_order=True).last() # pick up last record of previous date\n        test = test.join(lags_for_features.drop('time_id'), on=[\"date_id\", \"symbol_id\"],  how=\"left\")\n        #print(f'test, lags_ is not None: {test}')\n\n    \n    if days_passed > 5:\n        print(\"start retraining\")\n\n        ### GPT suggests resetting early stop each time model is retrained ###\n        early_stopping_rounds = 5\n        early_stop = xgb.callback.EarlyStopping(\n            rounds=early_stopping_rounds, save_best=True\n        )\n\n        #retrain_model = xgb.XGBRegressor(**XGB_Params)\n        #model = retrain_model.fit(X, y, sample_weight=w, xgb_model=model, eval_set=[(X[-1000000:], y[-1000000:])], \n        #                  sample_weight_eval_set=[w[-1000000:]], callbacks=[early_stop]) # .get_booster()\n        model.fit(X, y, sample_weight=w, eval_set=[(X[-1000000:], y[-1000000:])], \n                          sample_weight_eval_set=[w[-1000000:]], callbacks=[early_stop]) # .get_booster()\n        \n        #print('retraining finihsed')\n        days_passed = 0\n            \n        retrained = True\n\n        ### GPT suggests also resetting test after each model retrain ### \n        test = test.with_columns(\n                [pl.lit(0.0).cast(pl.Float32).alias(f'responder_{idx}_lag_1') for idx in range(9)]\n            )\n\n    else:\n        #print('else...')\n        test = test.with_columns(\n            [pl.lit(0.0).cast(pl.Float32).alias(f'responder_{idx}_lag_1') for idx in range(9)]\n        )\n        #print(f'test: {test}')\n    \n    #print(f'1. history_cache: {history_cache}')\n    history_cache.append(test.select(cols_needed_to_cache))\n        \n    X_test = test.select(feature_cols).fill_null(-1).to_numpy()\n    #print(f'2. X_test: {X_test}')\n    \n    preds = model.predict(X_test)\n    #print(f'3. preds: {preds}')\n    \n    # predictions = test.select('row_id').with_columns(\n    #     pl.Series(\n    #         name   = 'responder_6', \n    #         values = np.clip(preds, a_min = -5, a_max = 5),\n    #         dtype  = pl.Float64,\n    #     )\n    # )\n    \n    # Convert Polars DataFrame to Pandas DataFrame for prediction\n    # test_pandas = test.to_pandas()\n\n    # Extract the features for the model input\n    # responder_6_predictions = model.predict(test_pandas[features + dts])\n    \n    # Convert predictions to a Polars Series\n    predictions_series = pl.Series(\n        name='responder_6',\n        values=np.clip(preds, a_min=-5, a_max=5)\n    )\n    \n    # Add the predictions to the original Polars DataFrame\n    predictions = test.select('row_id').with_columns([predictions_series])\n    #print(f'4.predictions: {predictions}')\n\n    scoring = True\n        \n    # update time_step_count\n    time_step_count += 1\n    pbar.update(1)\n\n    end = time.time()\n\n    if retrained:\n        print(end - start)\n        \n    retrained = False\n\n    print('end of predict')\n    return predictions","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-27T17:43:48.633208Z","iopub.execute_input":"2024-12-27T17:43:48.633715Z","iopub.status.idle":"2024-12-27T17:43:48.656832Z","shell.execute_reply.started":"2024-12-27T17:43:48.633673Z","shell.execute_reply":"2024-12-27T17:43:48.655603Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pbar.refresh()\n\ninference_server = kaggle_evaluation.jane_street_inference_server.JSInferenceServer(predict)\nif os.getenv('KAGGLE_IS_COMPETITION_RERUN'):\n    #print('Server running')\n    inference_server.serve()\nelse:\n    #print('local gateway')\n    inference_server.run_local_gateway(\n        (\n            '/kaggle/input/jane-street-real-time-market-data-forecasting/test.parquet',\n            '/kaggle/input/jane-street-real-time-market-data-forecasting/lags.parquet',\n        )\n    )\n\npbar.close()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-27T17:45:04.639443Z","iopub.execute_input":"2024-12-27T17:45:04.640205Z","iopub.status.idle":"2024-12-27T17:45:04.702337Z","shell.execute_reply.started":"2024-12-27T17:45:04.640161Z","shell.execute_reply":"2024-12-27T17:45:04.701316Z"}},"outputs":[],"execution_count":null}]}