{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":84493,"databundleVersionId":9871156,"sourceType":"competition"},{"sourceId":213169289,"sourceType":"kernelVersion"},{"sourceId":202091,"sourceType":"modelInstanceVersion","isSourceIdPinned":true,"modelInstanceId":172415,"modelId":194756}],"dockerImageVersionId":30787,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# <span><h1 style = \"font-family: garamond; font-size: 40px; font-style: normal; letter-spcaing: 3px; background-color: #f6f5f5; color :#fe346e; border-radius: 100px 100px; text-align:center\">Import Libraries</h1></span>","metadata":{}},{"cell_type":"code","source":"source_file_path = '/kaggle/input/yunbase/Yunbase/baseline.py'\ntarget_file_path = '/kaggle/working/baseline.py'\nwith open(source_file_path, 'r', encoding='utf-8') as file:\n    content = file.read()\nwith open(target_file_path, 'w', encoding='utf-8') as file:\n    file.write(content)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T03:43:37.079036Z","iopub.execute_input":"2024-12-18T03:43:37.079876Z","iopub.status.idle":"2024-12-18T03:43:37.087036Z","shell.execute_reply.started":"2024-12-18T03:43:37.079843Z","shell.execute_reply":"2024-12-18T03:43:37.086212Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#!pip download -r Yunbase/requirements.txt","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-16T08:19:43.023055Z","iopub.execute_input":"2024-12-16T08:19:43.023377Z","iopub.status.idle":"2024-12-16T08:19:43.027304Z","shell.execute_reply.started":"2024-12-16T08:19:43.023349Z","shell.execute_reply":"2024-12-16T08:19:43.026404Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install -q --requirement /kaggle/input/yunbase/Yunbase/requirements.txt  \\\n--no-index --find-links file:/kaggle/input/yunbase/","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T03:43:45.809391Z","iopub.execute_input":"2024-12-18T03:43:45.810345Z","iopub.status.idle":"2024-12-18T03:44:01.649509Z","shell.execute_reply.started":"2024-12-18T03:43:45.810299Z","shell.execute_reply":"2024-12-18T03:44:01.648570Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from baseline import Yunbase\nimport polars as pl#similar to pandas, but with better performance when dealing with large datasets.\nimport pandas as pd#read csv,parquet\nimport numpy as np#for scientific computation of matrices\nimport joblib \n#model\nimport xgboost as xgb\nimport lightgbm as lgb\nimport catboost as cbt\nimport os#Libraries that interact with the operating system\nimport gc#rubbish collection\n#environment provided by competition hoster\nimport kaggle_evaluation.jane_street_inference_server\n\nimport random#provide some function to generate random_seed.\n#set random seed,to make sure model can be recurrented.\ndef seed_everything(seed):\n    np.random.seed(seed)#numpy's random seed\n    random.seed(seed)#python built-in random seed\nseed_everything(seed=2025)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T04:40:51.847368Z","iopub.execute_input":"2024-12-18T04:40:51.847750Z","iopub.status.idle":"2024-12-18T04:41:12.978531Z","shell.execute_reply.started":"2024-12-18T04:40:51.847713Z","shell.execute_reply":"2024-12-18T04:41:12.977544Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# <span><h1 style = \"font-family: garamond; font-size: 40px; font-style: normal; letter-spcaing: 3px; background-color: #f6f5f5; color :#fe346e; border-radius: 100px 100px; text-align:center\">Load Train Data</h1></span>\n\nI accidentally made a mistake by not filling in missing values in the training data and filling in -1 in the test data, resulting in a good LB (Those who know the reason can leave a message in the discussion forum)","metadata":{}},{"cell_type":"code","source":"yunbase=Yunbase()\ndata=[]\nfor i in [6,7,8,9]:\n    lazy_df = pl.scan_parquet(f\"/kaggle/input/jane-street-real-time-market-data-forecasting/train.parquet/partition_id={i}/part-0.parquet\")\n    df = lazy_df.collect()\n    df=df.to_pandas()\n    df['sin_time_id']=np.sin(2*np.pi*df['time_id']/967)\n    df['cos_time_id']=np.cos(2*np.pi*df['time_id']/967)\n    df['sin_time_id_halfday']=np.sin(2*np.pi*df['time_id']/483)\n    df['cos_time_id_halfday']=np.cos(2*np.pi*df['time_id']/483)\n    df=df.fillna(-1)\n    df=yunbase.reduce_mem_usage(df,float16_as32=False)\n    data.append(df)\ndf=pd.concat(data)\nprint(f\"df.shape:{df.shape}\")\ndel data\ngc.collect()#垃圾回收\nfinal_feature=['symbol_id','sin_time_id','cos_time_id','sin_time_id_halfday','cos_time_id_halfday']+[f'feature_0{i}' if i<10 else f'feature_{i}' for i in range(79)]\n#train=train[['responder_6']+final_feature]\ndf","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T04:41:16.220624Z","iopub.execute_input":"2024-12-18T04:41:16.221915Z","iopub.status.idle":"2024-12-18T04:42:32.184699Z","shell.execute_reply.started":"2024-12-18T04:41:16.221872Z","shell.execute_reply":"2024-12-18T04:42:32.183596Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Prepare training and validation data    \nnum_valid_dates = 180\ndatas = df['date_id'].unique()\nvalid_datas = datas[-num_valid_dates:]\ntrain_datas = datas[:-num_valid_dates]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T04:42:37.042502Z","iopub.execute_input":"2024-12-18T04:42:37.042867Z","iopub.status.idle":"2024-12-18T04:42:37.149893Z","shell.execute_reply.started":"2024-12-18T04:42:37.042839Z","shell.execute_reply":"2024-12-18T04:42:37.149101Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"valid_datas","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T04:42:40.003647Z","iopub.execute_input":"2024-12-18T04:42:40.004016Z","iopub.status.idle":"2024-12-18T04:42:40.011027Z","shell.execute_reply.started":"2024-12-18T04:42:40.003983Z","shell.execute_reply":"2024-12-18T04:42:40.010050Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **Define Customized Evaluation Method**\n\nwhich is R2 that specified by Jane Street","metadata":{}},{"cell_type":"code","source":"# Custom R2 metric for XGBoost\ndef r2_xgb(y_true, y_pred, sample_weight):\n    r2 = 1 - np.average((y_pred - y_true) ** 2, weights=sample_weight) / (np.average((y_true) ** 2, weights=sample_weight) + 1e-38)\n    return -r2\n\n# Custom R2 metric for LightGBM\ndef r2_lgb(y_true, y_pred, sample_weight):\n    r2 = 1 - np.average((y_pred - y_true) ** 2, weights=sample_weight) / (np.average((y_true) ** 2, weights=sample_weight) + 1e-38)\n    return 'r2', r2, True\n\n# Custom R2 metric for CatBoost\nclass r2_cbt(object):\n    def get_final_error(self, error, weight):\n        return 1 - error / (weight + 1e-38)\n\n    def is_max_optimal(self):\n        return True\n\n    def evaluate(self, approxes, target, weight):\n        assert len(approxes) == 1\n        assert len(target) == len(approxes[0])\n\n        approx = approxes[0]\n\n        error_sum = 0.0\n        weight_sum = 0.0\n\n        for i in range(len(approx)):\n            w = 1.0 if weight is None else weight[i]\n            weight_sum += w * (target[i] ** 2)\n            error_sum += w * ((approx[i] - target[i]) ** 2)\n\n        return error_sum, weight_sum","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T04:42:43.427243Z","iopub.execute_input":"2024-12-18T04:42:43.427607Z","iopub.status.idle":"2024-12-18T04:42:43.435319Z","shell.execute_reply.started":"2024-12-18T04:42:43.427576Z","shell.execute_reply":"2024-12-18T04:42:43.434419Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# <span><h1 style = \"font-family: garamond; font-size: 40px; font-style: normal; letter-spcaing: 3px; background-color: #f6f5f5; color :#fe346e; border-radius: 100px 100px; text-align:center\">Model training</h1></span>","metadata":{}},{"cell_type":"code","source":"lgb_params={\"boosting_type\": \"gbdt\",\"metric\": 'rmse',\n            'random_state': 2025,  \"max_depth\": 10,\"learning_rate\": 0.1,\n            \"n_estimators\": 120,\"colsample_bytree\": 0.6,\"colsample_bynode\": 0.6,\"verbose\": -1,\"reg_alpha\": 0.2,\n            \"reg_lambda\": 5,\"extra_trees\":True,'num_leaves':64,\"max_bin\":255,\n            'device':'gpu','gpu_use_dp':True,\n            }\n\ncat_params={'task_type':'GPU',\n           'random_state':2025,\n           'bagging_temperature' : 0.50,\n           'iterations'          : 200,\n           'learning_rate'       : 0.1,\n           'max_depth'           : 12,\n           'l2_leaf_reg'         : 1.25,\n           'min_data_in_leaf'    : 24,\n           'random_strength'     : 0.25, \n           'verbose'             : 0,\n          }\nxgb_params={'random_state': 2025, 'n_estimators': 125, \n            'learning_rate': 0.1, 'max_depth': 10,\n            'reg_alpha': 0.08, 'reg_lambda': 0.8, \n            'subsample': 0.95, 'colsample_bytree': 0.6, \n            'min_child_weight': 3,\n            'tree_method':'gpu_hist',\n           }\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T04:42:47.983982Z","iopub.execute_input":"2024-12-18T04:42:47.984926Z","iopub.status.idle":"2024-12-18T04:42:47.991316Z","shell.execute_reply.started":"2024-12-18T04:42:47.984886Z","shell.execute_reply":"2024-12-18T04:42:47.990039Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"if not os.path.exists(\"./models\"):\n    # Create the directory if it does not exist\n    os.mkdir(\"./models\")\nmodel_path = '/kaggle/input/baseline1/other/default/1'\nTRAINING = False\nif TRAINING:\n# Extract features, target, and weights for validation sets\n    X_valid = df[df['date_id'].isin(valid_datas)][final_feature].to_numpy()\n    y_valid = df[df['date_id'].isin(valid_datas)]['responder_6'].to_numpy().ravel()\n    w_valid = df[df['date_id'].isin(valid_datas)]['weight'].to_numpy().ravel()#扁平化\n# Extract features, target, and weights for training sets\n    X_train = df[df['date_id'].isin(train_datas)][final_feature].to_numpy()\n    y_train = df[df['date_id'].isin(train_datas)]['responder_6'].to_numpy().ravel()\n    w_train = df[df['date_id'].isin(train_datas)]['weight'].to_numpy().ravel()#扁平化\n# Initialize a list to store trained models\nmodels = []\n# Function to train a model or load a pre-trained model\ndef train(model_dict, model_name='lgb'):\n    if TRAINING:\n        # Select dates for training based on the fold number\n        #selected_dates = [date for ii, date in enumerate(train_datas) if ii % N_fold != i]\n        \n        # Get the model from the dictionary\n        model = model_dict[model_name]\n        # Train the model based on the type (LightGBM, XGBoost, or CatBoost)\n        if model_name == 'lgb':\n            # Train LightGBM model with early stopping and evaluation logging\n            model.fit(X_train, y_train, w_train,  \n                      eval_metric=[r2_lgb],\n                      eval_set=[(X_valid, y_valid, w_valid)], \n                      callbacks=[\n                          lgb.early_stopping(100), \n                          lgb.log_evaluation(10)\n                      ])\n            \n        elif model_name == 'cbt':\n            # Prepare evaluation set for CatBoost\n            evalset = cbt.Pool(X_valid, y_valid, weight=w_valid)\n            \n            # Train CatBoost model with early stopping and verbose logging\n            model.fit(X_train, y_train, sample_weight=w_train, \n                      eval_set=[evalset], \n                      verbose=10, \n                      early_stopping_rounds=100)\n            \n        else:\n            # Train XGBoost model with early stopping and verbose logging\n            model.fit(X_train, y_train, sample_weight=w_train, \n                      eval_set=[(X_valid, y_valid)], \n                      sample_weight_eval_set=[w_valid], \n                      verbose=10, \n                      early_stopping_rounds=100)\n\n        # Append the trained model to the list\n        models.append(model)\n        \n        # Save the trained model to a file\n        #joblib.dump(model, f'./models/{model_name}_{i}.model')\n        joblib.dump(model, f'./models/{model_name}.model')\n        \n        # Collect garbage to free up memory\n        import gc\n        gc.collect()\n        \n    else:\n        # If not in training mode, load the pre-trained model from the specified path\n        #models.append(joblib.load(f'{model_path}/{model_name}_{i}.model'))\n        models.append(joblib.load(f'{model_path}/{model_name}.model'))        \n    return \n    \n# Dictionary to store different models with their configurations\nmodel_dict = {\n    'lgb': lgb.LGBMRegressor(**lgb_params,objective='l2'),\n    'xgb': xgb.XGBRegressor(**xgb_params,eval_metric=r2_xgb,disable_default_eval_metric=True),\n    'cbt': cbt.CatBoostRegressor(**cat_params,eval_metric=r2_cbt()),\n}\n\n# Train models for each fold\n#for i in range(N_fold):\ntrain(model_dict, 'lgb')\ntrain(model_dict, 'xgb')\ntrain(model_dict, 'cbt')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T05:27:30.451617Z","iopub.execute_input":"2024-12-18T05:27:30.452025Z","iopub.status.idle":"2024-12-18T05:27:36.780355Z","shell.execute_reply.started":"2024-12-18T05:27:30.451990Z","shell.execute_reply":"2024-12-18T05:27:36.779581Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test = pl.scan_parquet(\"/kaggle/input/jane-street-real-time-market-data-forecasting/test.parquet/date_id=0/part-0.parquet\")\ntest = test.collect()\ntest = test.to_pandas()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T05:27:36.781746Z","iopub.execute_input":"2024-12-18T05:27:36.782032Z","iopub.status.idle":"2024-12-18T05:27:36.808470Z","shell.execute_reply.started":"2024-12-18T05:27:36.782004Z","shell.execute_reply":"2024-12-18T05:27:36.807809Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T05:27:43.975496Z","iopub.execute_input":"2024-12-18T05:27:43.975882Z","iopub.status.idle":"2024-12-18T05:27:43.998997Z","shell.execute_reply.started":"2024-12-18T05:27:43.975847Z","shell.execute_reply":"2024-12-18T05:27:43.998212Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test['sin_time_id']=np.sin(2*np.pi*test['time_id']/967)\ntest['cos_time_id']=np.cos(2*np.pi*test['time_id']/967)\ntest['sin_time_id_halfday']=np.sin(2*np.pi*test['time_id']/483)\ntest['cos_time_id_halfday']=np.cos(2*np.pi*test['time_id']/483)\ntest=test.fillna(-1)\ntest=test[final_feature]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T05:27:49.129667Z","iopub.execute_input":"2024-12-18T05:27:49.130420Z","iopub.status.idle":"2024-12-18T05:27:49.140405Z","shell.execute_reply.started":"2024-12-18T05:27:49.130385Z","shell.execute_reply":"2024-12-18T05:27:49.139466Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T01:02:55.072415Z","iopub.execute_input":"2024-12-18T01:02:55.073171Z","iopub.status.idle":"2024-12-18T01:02:55.099769Z","shell.execute_reply.started":"2024-12-18T01:02:55.073135Z","shell.execute_reply":"2024-12-18T01:02:55.098741Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test = test[final_feature].values  #加.values操作是为了将其转换为numpy数组形式","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T05:27:53.733577Z","iopub.execute_input":"2024-12-18T05:27:53.734344Z","iopub.status.idle":"2024-12-18T05:27:53.739999Z","shell.execute_reply.started":"2024-12-18T05:27:53.734307Z","shell.execute_reply":"2024-12-18T05:27:53.738823Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pred = [model.predict(test) for model in models]\npred = pred[0]*0.55+pred[1]*0.25+pred[2]*0.2\n#pred = np.mean(pred, axis=0)\npred","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T05:27:55.558779Z","iopub.execute_input":"2024-12-18T05:27:55.559521Z","iopub.status.idle":"2024-12-18T05:27:55.607915Z","shell.execute_reply.started":"2024-12-18T05:27:55.559482Z","shell.execute_reply":"2024-12-18T05:27:55.607049Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# debug the submission","metadata":{}},{"cell_type":"code","source":"test = pl.scan_parquet(\"/kaggle/input/jane-street-real-time-market-data-forecasting/test.parquet/date_id=0/part-0.parquet\")\ntest = test.collect()\npredictions = test.select(\n        'row_id',\n        pl.lit(0.0).alias('responder_6'),\n    )\ntest=test.to_pandas()\ntest['sin_time_id']=np.sin(2*np.pi*test['time_id']/967)\ntest['cos_time_id']=np.cos(2*np.pi*test['time_id']/967)\ntest['sin_time_id_halfday']=np.sin(2*np.pi*test['time_id']/483)\ntest['cos_time_id_halfday']=np.cos(2*np.pi*test['time_id']/483)\ntest=test.fillna(-1)\ntest=test[final_feature]\neps=1e-10\ntest_preds=[model.predict(test) for model in models]\n#test_preds = np.mean(test_preds, axis=0)\ntest_preds = test_preds[0]*0.55+test_preds[1]*0.25+test_preds[2]*0.2\ntest_preds=np.clip(test_preds,-5+eps,5-eps)\npredictions = predictions.with_columns(pl.Series('responder_6', test_preds.ravel()))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T05:27:59.483847Z","iopub.execute_input":"2024-12-18T05:27:59.484611Z","iopub.status.idle":"2024-12-18T05:27:59.565688Z","shell.execute_reply.started":"2024-12-18T05:27:59.484574Z","shell.execute_reply":"2024-12-18T05:27:59.564913Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"predictions.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T05:28:02.946715Z","iopub.execute_input":"2024-12-18T05:28:02.947596Z","iopub.status.idle":"2024-12-18T05:28:02.961067Z","shell.execute_reply.started":"2024-12-18T05:28:02.947555Z","shell.execute_reply":"2024-12-18T05:28:02.960267Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# <span><h1 style = \"font-family: garamond; font-size: 40px; font-style: normal; letter-spcaing: 3px; background-color: #f6f5f5; color :#fe346e; border-radius: 100px 100px; text-align:center\">Model Inference</h1></span>\n\n#### The following code is used in the training set to load more data for training, and not in the testing set to save time. I tried calling this function 100000 times and it would take about an hour.\n\n```python\ntest=yunbase.reduce_mem_usage(test,float16_as32=False)\n```","metadata":{}},{"cell_type":"code","source":"# Global lags storage\nlags_: pl.DataFrame | None = None\ndef predict(test: pl.DataFrame, lags: pl.DataFrame | None) -> pl.DataFrame:\n    global lags_\n\n# Logic for saving or loading lags\n    if lags is not None:\n        lags_ = lags\n        \n    predictions = test.select(\n        'row_id',\n        pl.lit(0.0).alias('responder_6'),\n    )\n    test=test.to_pandas()\n    test['sin_time_id']=np.sin(2*np.pi*test['time_id']/967)\n    test['cos_time_id']=np.cos(2*np.pi*test['time_id']/967)\n    test['sin_time_id_halfday']=np.sin(2*np.pi*test['time_id']/483)\n    test['cos_time_id_halfday']=np.cos(2*np.pi*test['time_id']/483)\n    test=test.fillna(-1)\n    test=test[final_feature]\n    eps=1e-10\n    test_preds=[model.predict(test) for model in models]\n    #test_preds = np.mean(test_preds, axis=0)\n    test_preds = test_preds[0]*0.55+test_preds[1]*0.25+test_preds[2]*0.2\n    test_preds=np.clip(test_preds,-5+eps,5-eps)\n    predictions = predictions.with_columns(pl.Series('responder_6', test_preds.ravel()))\n    return predictions","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T05:28:29.148016Z","iopub.execute_input":"2024-12-18T05:28:29.148841Z","iopub.status.idle":"2024-12-18T05:28:29.156000Z","shell.execute_reply.started":"2024-12-18T05:28:29.148802Z","shell.execute_reply":"2024-12-18T05:28:29.155019Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time\ninference_server = kaggle_evaluation.jane_street_inference_server.JSInferenceServer(predict)\n\nif os.getenv('KAGGLE_IS_COMPETITION_RERUN'):\n    inference_server.serve()\nelse:\n    inference_server.run_local_gateway(\n        (\n            '/kaggle/input/jane-street-real-time-market-data-forecasting/test.parquet',\n            '/kaggle/input/jane-street-real-time-market-data-forecasting/lags.parquet',\n        )\n    )","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T05:28:34.110007Z","iopub.execute_input":"2024-12-18T05:28:34.110779Z","iopub.status.idle":"2024-12-18T05:28:34.288598Z","shell.execute_reply.started":"2024-12-18T05:28:34.110739Z","shell.execute_reply":"2024-12-18T05:28:34.287538Z"}},"outputs":[],"execution_count":null}]}