{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":84493,"databundleVersionId":9871156,"sourceType":"competition"},{"sourceId":203900450,"sourceType":"kernelVersion"}],"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true},"papermill":{"default_parameters":{},"duration":7.594014,"end_time":"2024-10-10T11:58:36.355301","environment_variables":{},"exception":null,"input_path":"__notebook__.ipynb","output_path":"__notebook__.ipynb","parameters":{},"start_time":"2024-10-10T11:58:28.761287","version":"2.6.0"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"**Note**  \n**Training Notebook cannot run at kaggle platform (because many memory is requied to run).**  \n**If you want to execute this code, you need to prepare own computations (out of kaggle).**  \n\n# Baseline notebooks:\n- Preprocessing : https://www.kaggle.com/code/motono0223/js24-preprocessing-create-lags\n- Training (Code only) : **this notebook** https://www.kaggle.com/code/motono0223/js24-train-gbdt-model-with-lags-singlemodel\n  - trained model : https://www.kaggle.com/datasets/motono0223/js24-trained-gbdt-model\n- Inference : https://www.kaggle.com/code/motono0223/js24-inference-gbdt-with-lags-singlemodel\n- EDA(1) : https://www.kaggle.com/code/motono0223/eda-jane-street-real-time-market-data-forecasting\n- EDA(2) : https://www.kaggle.com/code/motono0223/eda-v2-jane-street-real-time-market-forecasting","metadata":{}},{"cell_type":"code","source":"# # This Python 3 environment comes with many helpful analytics libraries installed\n# # It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# # For example, here's several helpful packages to load\n\n# import numpy as np # linear algebra\n# import pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# # Input data files are available in the read-only \"../input/\" directory\n# # For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\n# import os\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n#         print(os.path.join(dirname, filename))\n\n# # You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a ver","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T16:29:58.569053Z","iopub.execute_input":"2025-01-01T16:29:58.569454Z","iopub.status.idle":"2025-01-01T16:29:59.230137Z","shell.execute_reply.started":"2025-01-01T16:29:58.569422Z","shell.execute_reply":"2025-01-01T16:29:59.229329Z"},"collapsed":true,"jupyter":{"outputs_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# import pandas as pd\n# import polars as pl\n# import numpy as np\n# import os\n# from tqdm.auto import tqdm\n# from matplotlib import pyplot as plt\n# import pickle\n\n# from sklearn.metrics import r2_score\n# from lightgbm import LGBMRegressor\n# import lightgbm as lgb\n# from xgboost import XGBRegressor\n# from catboost import CatBoostRegressor\n# from sklearn.ensemble import VotingRegressor\n\n# import warnings\n# warnings.filterwarnings('ignore')\n# pd.options.display.max_columns = None\n\n# import kaggle_evaluation.jane_street_inference_server","metadata":{"execution":{"iopub.status.busy":"2025-01-05T05:06:50.988171Z","iopub.execute_input":"2025-01-05T05:06:50.988409Z","iopub.status.idle":"2025-01-05T05:06:56.031652Z","shell.execute_reply.started":"2025-01-05T05:06:50.988387Z","shell.execute_reply":"2025-01-05T05:06:56.030432Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Configurations","metadata":{}},{"cell_type":"code","source":"class CONFIG:\n    seed = 2025\n    target_col = \"responder_6\"\n    feature_cols = [\"symbol_id\", \"time_id\"] \\\n        + [f\"feature_{idx:02d}\" for idx in range(79)] \\\n        + [f\"responder_{idx}_lag_1\" for idx in range(9)]\n    categorical_cols = []","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-09T13:56:02.180527Z","iopub.execute_input":"2025-01-09T13:56:02.180852Z","iopub.status.idle":"2025-01-09T13:56:02.185385Z","shell.execute_reply.started":"2025-01-09T13:56:02.180830Z","shell.execute_reply":"2025-01-09T13:56:02.184431Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Load Data","metadata":{}},{"cell_type":"code","source":"# import dask.dataframe as dd\n# from pathlib import Path\n# import pandas as pd\n\n# # 构造一个包含所有目标 date_id 分区的路径列表\n# base_path = \"/kaggle/input/js24-preprocessing-create-lags/training.parquet\"\n# date_ids = range(1200, 1601)  # 包含1600所以要加1\n# parquet_paths = [str(Path(base_path) / f\"date_id={date_id}\" / \"00000000.parquet\") for date_id in date_ids]\n\n# # 创建一个空的 Dask DataFrame 列表用于存储转换后的数据框\n# dfs = []\n\n# # 循环遍历每个路径，并使用 Pandas 读取和转换类型\n# for path in parquet_paths:\n#     df_pandas = pd.read_parquet(path)\n#     # 确保 date_id 是 int32 类型\n#     df_pandas['date_id'] = df_pandas['date_id'].astype('int16')\n#     # 将 Pandas DataFrame 转换为 Dask DataFrame 并添加到列表中\n#     dfs.append(dd.from_pandas(df_pandas, npartitions=1))\n\n# # 合并所有的 Dask DataFrames\n# df = dd.concat(dfs)\n\n# # 查看数据框前几行\n# print(df.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-05T09:23:11.706610Z","iopub.execute_input":"2025-01-05T09:23:11.706991Z","iopub.status.idle":"2025-01-05T09:24:04.386550Z","shell.execute_reply.started":"2025-01-05T09:23:11.706948Z","shell.execute_reply":"2025-01-05T09:24:04.385460Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# df_pandas=df.compute()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-05T09:31:54.173820Z","iopub.execute_input":"2025-01-05T09:31:54.174269Z","iopub.status.idle":"2025-01-05T09:32:01.851641Z","shell.execute_reply.started":"2025-01-05T09:31:54.174234Z","shell.execute_reply":"2025-01-05T09:32:01.850407Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import polars as pl\ntrain = pl.scan_parquet(\"/kaggle/input/js24-preprocessing-create-lags/training.parquet\").collect().to_pandas()\n# train=df_pandas\ntrain.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-09T13:52:37.703737Z","iopub.execute_input":"2025-01-09T13:52:37.704078Z","iopub.status.idle":"2025-01-09T13:53:19.927185Z","shell.execute_reply.started":"2025-01-09T13:52:37.704051Z","shell.execute_reply":"2025-01-09T13:53:19.926231Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train1=train[train['date_id']>1300]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-09T13:54:08.623324Z","iopub.execute_input":"2025-01-09T13:54:08.623860Z","iopub.status.idle":"2025-01-09T13:54:13.633563Z","shell.execute_reply.started":"2025-01-09T13:54:08.623830Z","shell.execute_reply":"2025-01-09T13:54:13.632789Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import gc\ndel train\ngc.collect()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-09T13:54:16.093888Z","iopub.execute_input":"2025-01-09T13:54:16.094187Z","iopub.status.idle":"2025-01-09T13:54:16.925425Z","shell.execute_reply.started":"2025-01-09T13:54:16.094168Z","shell.execute_reply":"2025-01-09T13:54:16.924556Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"valid = pl.scan_parquet(\"/kaggle/input/js24-preprocessing-create-lags/validation.parquet\").collect().to_pandas()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-09T13:54:21.988077Z","iopub.execute_input":"2025-01-09T13:54:21.988475Z","iopub.status.idle":"2025-01-09T13:54:24.173170Z","shell.execute_reply.started":"2025-01-09T13:54:21.988446Z","shell.execute_reply":"2025-01-09T13:54:24.172380Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\n# 合并 train 和 valid\ncombined = pd.concat([train1, valid])\n\n# 删除原始的 train 和 valid 以释放内存\ndel train1\ngc.collect()\n\n# 重置索引\ncombined.reset_index(drop=True, inplace=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-09T13:54:26.978728Z","iopub.execute_input":"2025-01-09T13:54:26.979074Z","iopub.status.idle":"2025-01-09T13:54:30.132150Z","shell.execute_reply.started":"2025-01-09T13:54:26.979048Z","shell.execute_reply":"2025-01-09T13:54:30.131129Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"combined.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-09T08:33:33.252571Z","iopub.execute_input":"2025-01-09T08:33:33.252934Z","iopub.status.idle":"2025-01-09T08:33:33.257671Z","shell.execute_reply.started":"2025-01-09T08:33:33.252902Z","shell.execute_reply":"2025-01-09T08:33:33.256952Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# GBDT models","metadata":{}},{"cell_type":"code","source":"# def get_model(seed):\n#     # XGBoost parameters\n#     XGB_Params = {\n#         'learning_rate': 0.05,\n#         'max_depth': 6,\n#         'n_estimators': 200,\n#         'subsample': 0.8,\n#         'colsample_bytree': 0.8,\n#         'reg_alpha': 1,\n#         'reg_lambda': 5,\n#         'random_state': seed,\n#         'tree_method': 'gpu_hist',\n#         'device' : 'cuda',\n#         'n_gpus' : 2,\n#     }\n    \n#     XGB_Model = XGBRegressor(**XGB_Params)\n#     return XGB_Model","metadata":{"execution":{"iopub.status.busy":"2025-01-03T14:55:57.552286Z","iopub.execute_input":"2025-01-03T14:55:57.552723Z","iopub.status.idle":"2025-01-03T14:55:57.559304Z","shell.execute_reply.started":"2025-01-03T14:55:57.552682Z","shell.execute_reply":"2025-01-03T14:55:57.558031Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Training model","metadata":{}},{"cell_type":"code","source":"# import pandas as pd\n# import gc\n\n# # 打印当前内存使用情况\n# def print_memory_usage():\n#     import psutil\n#     process = psutil.Process()\n#     print(f\"Memory usage: {process.memory_info().rss / 1024 ** 2:.2f} MB\")\n\n# print(\"Before slicing:\")\n# print_memory_usage()\n\n# # 使用 .loc 或 .iloc 来避免隐式复制\n# X_train = combined.loc[:, CONFIG.feature_cols].copy()\n# y_train = combined.loc[:, [CONFIG.target_col]].squeeze().copy()  # 确保是Series\n# w_train = combined.loc[:, [\"weight\"]].squeeze().copy()\n\n# X_valid = valid.loc[:, CONFIG.feature_cols].copy()\n# y_valid = valid.loc[:, [CONFIG.target_col]].squeeze().copy()\n# w_valid = valid.loc[:, [\"weight\"]].squeeze().copy()\n\n# print(\"After slicing, before deletion:\")\n# print_memory_usage()\n\n# # 删除原始DataFrame并强制垃圾回收\n# del valid, combined\n# gc.collect()\n\n# print(\"After deletion and garbage collection:\")\n# print_memory_usage()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-09T13:51:38.312448Z","iopub.execute_input":"2025-01-09T13:51:38.312754Z","execution_failed":"2025-01-09T13:52:06.126Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_train = combined[ CONFIG.feature_cols ]\ny_train = combined[ CONFIG.target_col ]\nw_train = combined[ \"weight\" ]\nX_valid = valid[ CONFIG.feature_cols ]\ny_valid = valid[ CONFIG.target_col ]\nw_valid = valid[ \"weight\" ]\n\nX_train.shape, y_train.shape, w_train.shape, X_valid.shape, y_valid.shape, w_valid.shape","metadata":{"execution":{"iopub.status.busy":"2025-01-09T13:56:12.896072Z","iopub.execute_input":"2025-01-09T13:56:12.896384Z","iopub.status.idle":"2025-01-09T13:56:15.375226Z","shell.execute_reply.started":"2025-01-09T13:56:12.896357Z","shell.execute_reply":"2025-01-09T13:56:15.374416Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"del valid,combined\ngc.collect()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-09T13:56:20.181227Z","iopub.execute_input":"2025-01-09T13:56:20.181550Z","iopub.status.idle":"2025-01-09T13:56:20.241492Z","shell.execute_reply.started":"2025-01-09T13:56:20.181528Z","shell.execute_reply":"2025-01-09T13:56:20.240681Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import lightgbm as lgb\n# lgb_params={\"boosting_type\": \"gbdt\",\"metric\": 'rmse',\n#             'random_state': 2025,  \"max_depth\": 10,\"learning_rate\": 0.1,\n#             \"n_estimators\": 120,\"colsample_bytree\": 0.6,\"colsample_bynode\": 0.6,\"verbose\": -1,\"reg_alpha\": 0.2,\n#             \"reg_lambda\": 5,\"extra_trees\":True,'num_leaves':64,\"max_bin\":255,\n#             'device':'gpu','gpu_use_dp':True,\n#             }\n\n# lgb.LGBMRegressor(n_estimators=500, device='gpu', gpu_use_dp=True, objective='l2')\n\n# def get_model(seed):\n#     # XGBoost parameters\n#     XGB_Params = {\n#         'learning_rate': 0.05,\n#         'max_depth': 6,\n#         'n_estimators': 200,\n#         'subsample': 0.8,\n#         'colsample_bytree': 0.8,\n#         'reg_alpha': 1,\n#         'reg_lambda': 5,\n#         'random_state': seed,\n#         'tree_method': 'gpu_hist',\n#         'device' : 'cuda',\n#         'n_gpus' : 2,\n#     }\n    \n#     XGB_Model = XGBRegressor(**XGB_Params)\n#     return XGB_Model\n\ndef get_model(seed):\n    # LightGBM parameters\n    LGBM_Params = {\n        'learning_rate': 0.1,\n        'max_depth': 10,  # LightGBM 默认是不限制深度\n        'num_leaves': 63,  # 相当于 XGBoost 的 max_depth 参数\n        'n_estimators': 200,\n        # 'subsample': 0.8,\n        'colsample_bytree': 0.7,\n        # 'reg_alpha': 1,\n        'reg_lambda': 5,\n        'random_state': seed,\n        'device': 'gpu',  # 使用 GPU 加速\n        # 'gpu_platform_id': 0,  # 根据需要设置\n        # 'gpu_device_id': 0,  # 根据需要设置\n        'metric': 'rmse',  # 根据任务选择合适的评估指标\n        'gpu_use_dp':True\n    }\n    \n    return LGBM_Params\n\n# 准备数据集\ntrain_data = lgb.Dataset(X_train, label=y_train, weight=w_train)\nvalid_data = lgb.Dataset(X_valid, label=y_valid, weight=w_valid, reference=train_data)\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-09T13:56:27.893299Z","iopub.execute_input":"2025-01-09T13:56:27.893673Z","iopub.status.idle":"2025-01-09T13:56:32.441484Z","shell.execute_reply.started":"2025-01-09T13:56:27.893619Z","shell.execute_reply":"2025-01-09T13:56:32.440537Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# #train xgb\n# %%time\n# model = get_model(CONFIG.seed)\n# model.fit( X_train, y_train, sample_weight=w_train)","metadata":{"execution":{"iopub.status.busy":"2025-01-03T14:58:18.556271Z","iopub.execute_input":"2025-01-03T14:58:18.556696Z","execution_failed":"2025-01-03T15:00:16.395Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 获取模型参数\nparams = get_model(CONFIG.seed)\n\n# 创建回调函数列表\ncallbacks = [\n    lgb.early_stopping(stopping_rounds=50, verbose=True),\n    lgb.log_evaluation(period=100)  # 可选：调整日志输出频率\n]\n\n# 训练模型\nmodel = lgb.train(\n    params,\n    train_data,\n    valid_sets=[valid_data],\n    valid_names=['valid'],\n    callbacks=callbacks  # 添加回调函数\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-09T13:57:00.493800Z","iopub.execute_input":"2025-01-09T13:57:00.494436Z","iopub.status.idle":"2025-01-09T14:00:58.305923Z","shell.execute_reply.started":"2025-01-09T13:57:00.494407Z","shell.execute_reply":"2025-01-09T14:00:58.305165Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y_pred_train1 = model.predict(X_train.iloc[:X_train.shape[0]//2])\ny_pred_train2 = model.predict(X_train.iloc[X_train.shape[0]//2:])\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-09T14:04:49.466168Z","iopub.execute_input":"2025-01-09T14:04:49.466585Z","iopub.status.idle":"2025-01-09T14:08:21.057068Z","shell.execute_reply.started":"2025-01-09T14:04:49.466560Z","shell.execute_reply":"2025-01-09T14:08:21.056077Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nfrom sklearn.metrics import r2_score\ntrain_score = r2_score(y_train, np.concatenate([y_pred_train1, y_pred_train2], axis=0), sample_weight=w_train )\ntrain_score","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-09T14:08:39.337835Z","iopub.execute_input":"2025-01-09T14:08:39.338203Z","iopub.status.idle":"2025-01-09T14:08:39.633496Z","shell.execute_reply.started":"2025-01-09T14:08:39.338176Z","shell.execute_reply":"2025-01-09T14:08:39.632654Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y_pred_valid = model.predict(X_valid)\nvalid_score = r2_score(y_valid, y_pred_valid, sample_weight=w_valid )\nvalid_score","metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","papermill":{"duration":1.594977,"end_time":"2024-10-10T11:58:33.569648","exception":false,"start_time":"2024-10-10T11:58:31.974671","status":"completed"},"tags":[],"trusted":true,"execution":{"iopub.status.busy":"2025-01-09T14:08:49.427528Z","iopub.execute_input":"2025-01-09T14:08:49.427917Z","iopub.status.idle":"2025-01-09T14:09:01.522266Z","shell.execute_reply.started":"2025-01-09T14:08:49.427889Z","shell.execute_reply":"2025-01-09T14:09:01.521325Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# y_means = { symbol_id : -1 for symbol_id in range(39) }\n# for symbol_id, gdf in train[[\"symbol_id\", CONFIG.target_col]].groupby(\"symbol_id\"):\n#     y_mean = gdf[ CONFIG.target_col ].mean()\n#     y_means[symbol_id] = y_mean\n#     print(f\"symbol_id = {symbol_id}, y_means = {y_mean:.5f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-05T10:06:20.809535Z","iopub.execute_input":"2025-01-05T10:06:20.809829Z","iopub.status.idle":"2025-01-05T10:06:20.823972Z","shell.execute_reply.started":"2025-01-05T10:06:20.809809Z","shell.execute_reply":"2025-01-05T10:06:20.822986Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# cv_detail = { symbol_id : 0 for symbol_id in range(39) }\n# for symbol_id, gdf in valid.groupby(\"symbol_id\"):\n#     X_valid = gdf[ CONFIG.feature_cols ]\n#     y_valid = gdf[ CONFIG.target_col ]\n#     w_valid = gdf[ \"weight\" ]\n#     y_pred_valid = model.predict(X_valid)\n#     score = r2_score(y_valid, y_pred_valid, sample_weight=w_valid )\n#     cv_detail[symbol_id] = score\n    \n#     print(f\"symbol_id = {symbol_id}, score = {score:.5f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-05T10:06:42.394214Z","iopub.execute_input":"2025-01-05T10:06:42.394591Z","iopub.status.idle":"2025-01-05T10:06:42.408925Z","shell.execute_reply.started":"2025-01-05T10:06:42.394562Z","shell.execute_reply":"2025-01-05T10:06:42.407761Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# sids = list(cv_detail.keys())\n# plt.bar(sids, [cv_detail[sid] for sid in sids])\n# plt.grid()\n# plt.xlabel(\"symbol_id\")\n# plt.ylabel(\"CV score\")\n# plt.show()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# import lightgbm as lgb\n\n# def get_model(seed):\n#     # LightGBM parameters\n#     LGBM_Params = {\n#         'learning_rate': 0.05,\n#         'max_depth': 6,\n#         'n_estimators': 200,\n#         'subsample': 0.8,\n#         'colsample_bytree': 0.8,\n#         'reg_alpha': 1,\n#         'reg_lambda': 5,\n#         'random_state': seed,\n#         'device': 'gpu',  # Use GPU if available\n#         'objective': 'regression',  # Specify the objective for regression task\n#     }\n    \n#     LGBM_Model = lgb.LGBMRegressor(**LGBM_Params)\n#     return LGBM_Model\n\n# # Assuming you have defined X_train, y_train, w_train, X_valid, y_valid, and w_valid as before\n\n# %%time\n# model = get_model(CONFIG.seed)\n\n# # Train the model with sample weights\n# model.fit(X_train, y_train, sample_weight=w_train, eval_set=[(X_valid, y_valid)], eval_sample_weight=[w_valid], eval_metric='rmse', verbose=10)\n\n# # Predictions and evaluation remain the same\n# y_pred_train1 = model.predict(X_train.iloc[:X_train.shape[0]//2])\n# y_pred_train2 = model.predict(X_train.iloc[X_train.shape[0]//2:])\n# train_score = r2_score(y_train, np.concatenate([y_pred_train1, y_pred_train2], axis=0), sample_weight=w_train)\n# print(f'Train score: {train_score}')\n\n# y_pred_valid = model.predict(X_valid)\n# valid_score = r2_score(y_valid, y_pred_valid, sample_weight=w_valid)\n# print(f'Validation score: {valid_score}')\n\n# # The rest of the code remains unchanged","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T16:13:30.191271Z","iopub.execute_input":"2025-01-01T16:13:30.191606Z","iopub.status.idle":"2025-01-01T16:13:33.999578Z","shell.execute_reply.started":"2025-01-01T16:13:30.191563Z","shell.execute_reply":"2025-01-01T16:13:33.997946Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Save result","metadata":{}},{"cell_type":"code","source":"import pickle as pkl\nresult = {\n    \"model\" : model,\n    \"cv\" : valid_score,\n    \"cv_detail\" : -1,\n    \"y_mean\" : -1,\n}\nwith open(\"result_lgbm3.pkl\", \"wb\") as fp:\n    pkl.dump(result, fp)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-09T14:09:20.603096Z","iopub.execute_input":"2025-01-09T14:09:20.603462Z","iopub.status.idle":"2025-01-09T14:09:20.671403Z","shell.execute_reply.started":"2025-01-09T14:09:20.603429Z","shell.execute_reply":"2025-01-09T14:09:20.670689Z"}},"outputs":[],"execution_count":null}]}