{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84493,"databundleVersionId":9849268,"sourceType":"competition"}],"dockerImageVersionId":30786,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Overview\nThis repository provides code implementation for training Gradient Boosting Models (GBMs), a popular machine learning technique for both classification and regression tasks. GBMs are ensemble methods that combine the predictions of several base estimators to improve accuracy and generalization performance.\n\n","metadata":{}},{"cell_type":"markdown","source":"# Inference\n[[JSR-TMDF] Gradient Boosting Models (Inference)](https://www.kaggle.com/code/takaito/jsr-tmdf-gradient-boosting-models-inference)","metadata":{}},{"cell_type":"markdown","source":"# Tips\n## 1. CV Strategy\nBy setting kfold = KFold(n_splits=CFG.N_SPLIT, shuffle=False), the data is being loaded in chronological order, so the splitting is performed based on the time series.\n\n## 2. feature importance\nIn LightGBM, we save the feature importance. This allows you to check which features are effective and can provide insights for removing unnecessary features or creating new ones, so please make use of it.","metadata":{}},{"cell_type":"markdown","source":"To be updated!! (I plan to add more hints if the number of votes increases.)","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ====================================================\n# Library\n# ====================================================\nimport os\nimport gc\nimport warnings\nwarnings.filterwarnings('ignore')\nimport random\nimport scipy as sp\nimport numpy as np\nimport pandas as pd\nimport polars as pl\nfrom glob import glob\nfrom pathlib import Path\nimport joblib\nimport pickle\nimport itertools\nfrom tqdm.auto import tqdm\n\nimport torch\nfrom sklearn.model_selection import KFold, StratifiedKFold, train_test_split, GroupKFold\nfrom sklearn.metrics import log_loss, roc_auc_score, matthews_corrcoef, f1_score\nfrom sklearn.metrics import mean_squared_error, r2_score\nfrom sklearn.preprocessing import LabelEncoder\nimport lightgbm as lgb\nimport xgboost as xgb\nfrom catboost import Pool, CatBoostRegressor, CatBoostClassifier","metadata":{"execution":{"iopub.status.busy":"2024-10-14T20:59:00.401302Z","iopub.execute_input":"2024-10-14T20:59:00.401779Z","iopub.status.idle":"2024-10-14T20:59:00.412134Z","shell.execute_reply.started":"2024-10-14T20:59:00.401734Z","shell.execute_reply":"2024-10-14T20:59:00.410892Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!mkdir oof\n!mkdir models","metadata":{"execution":{"iopub.status.busy":"2024-10-14T20:59:00.414621Z","iopub.execute_input":"2024-10-14T20:59:00.415093Z","iopub.status.idle":"2024-10-14T20:59:02.877612Z","shell.execute_reply.started":"2024-10-14T20:59:00.415050Z","shell.execute_reply":"2024-10-14T20:59:02.875967Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ====================================================\n# Configurations\n# ====================================================\nclass CFG:\n    VER = 1\n    AUTHOR = 'takaito'\n    COMPETITION = 'jane-street-real-time-market-data-forecasting'\n    DATA_PATH = Path('/kaggle/input/jane-street-real-time-market-data-forecasting')\n    OOF_DATA_PATH = Path('./oof')\n    MODEL_DATA_PATH = Path('./models')\n    METHOD_LIST = ['lightgbm', 'xgboost', 'catboost']\n    USE_GPU = torch.cuda.is_available()\n    SEED = 42\n    N_SPLIT = 5\n    target_col = 'responder_6'\n    metric = 'r2_score'\n    metric_maximize_flag = True\n\n    num_boost_round = 2500\n    early_stopping_round = 10\n    verbose = 50\n    \n    regression_lgb_params = {\n        'objective': 'regression',\n        'metric': 'rmse', \n        'learning_rate': 0.05,\n        'num_leaves': 31,\n        'seed': SEED,\n    }\n    regression_xgb_params = {\n        'objective': 'reg:squarederror',\n        'eval_metric': 'rmse',\n        'learning_rate': 0.05, \n        'max_depth': 7,\n        'random_state': SEED,\n    }\n    \n    regression_cat_params = {\n        'loss_function': 'RMSE',\n        'learning_rate': 0.05, \n        'iterations': num_boost_round, \n        'depth': 7, \n        'random_seed': SEED,\n    }\n    ","metadata":{"execution":{"iopub.status.busy":"2024-10-14T20:59:02.879578Z","iopub.execute_input":"2024-10-14T20:59:02.880036Z","iopub.status.idle":"2024-10-14T20:59:02.891902Z","shell.execute_reply.started":"2024-10-14T20:59:02.879984Z","shell.execute_reply":"2024-10-14T20:59:02.890491Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ====================================================\n# Seed everything\n# ====================================================\ndef seed_everything(seed):\n    random.seed(seed)\n    np.random.seed(seed)\n    os.environ['PYTHONHASHSEED'] = str(seed)\nseed_everything(CFG.SEED)","metadata":{"execution":{"iopub.status.busy":"2024-10-14T20:59:02.895360Z","iopub.execute_input":"2024-10-14T20:59:02.896738Z","iopub.status.idle":"2024-10-14T20:59:02.909467Z","shell.execute_reply.started":"2024-10-14T20:59:02.896687Z","shell.execute_reply":"2024-10-14T20:59:02.907989Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def lightgbm_training(x_train: pd.DataFrame, y_train: pd.DataFrame, x_valid: pd.DataFrame, y_valid: pd.DataFrame):\n    lgb_train = lgb.Dataset(x_train, y_train)\n    lgb_valid = lgb.Dataset(x_valid, y_valid)\n    \n    model = lgb.train(\n                params = CFG.regression_lgb_params,\n                train_set = lgb_train,\n                num_boost_round = CFG.num_boost_round,\n                valid_sets = [lgb_train, lgb_valid],\n                callbacks=[lgb.early_stopping(stopping_rounds=CFG.early_stopping_round, verbose=CFG.verbose),\n                           lgb.log_evaluation(CFG.verbose),\n                          ]\n            )\n    # Predict validation\n    valid_pred = model.predict(x_valid)\n    return model, valid_pred\ndef xgboost_training(x_train: pd.DataFrame, y_train: pd.DataFrame, x_valid: pd.DataFrame, y_valid: pd.DataFrame):\n    xgb_train = xgb.DMatrix(data=x_train, label=y_train)\n    xgb_valid = xgb.DMatrix(data=x_valid, label=y_valid)\n    model = xgb.train(\n                CFG.regression_xgb_params,\n                dtrain = xgb_train,\n                num_boost_round = CFG.num_boost_round,\n                evals = [(xgb_train, 'train'), (xgb_valid, 'eval')],\n                early_stopping_rounds = CFG.early_stopping_round,\n                verbose_eval = CFG.verbose\n            )\n    # Predict validation\n    valid_pred = model.predict(xgb.DMatrix(x_valid))\n    return model, valid_pred\ndef catboost_training(x_train: pd.DataFrame, y_train: pd.DataFrame, x_valid: pd.DataFrame, y_valid: pd.DataFrame):\n    cat_train = Pool(data=x_train, label=y_train)\n    cat_valid = Pool(data=x_valid, label=y_valid)\n    model = CatBoostRegressor(**CFG.regression_cat_params)\n    model.fit(cat_train,\n              eval_set = [cat_valid],\n              early_stopping_rounds = CFG.early_stopping_round,\n              verbose = CFG.verbose,\n              use_best_model = True)\n    # Predict validation\n    valid_pred = model.predict(x_valid)\n    return model, valid_pred\n\ndef gradient_boosting_model_cv_training(method: str, train_df: pd.DataFrame, features: list):\n    # Create a numpy array to store out of folds predictions\n    oof_predictions = np.zeros(len(train_df))\n    oof_fold = np.zeros(len(train_df))\n    ## 1. CV Strategy\n    kfold = KFold(n_splits=CFG.N_SPLIT, shuffle=False) # , shuffle=True, random_state=CFG.SEED)\n    for fold, (train_index, valid_index) in enumerate(kfold.split(X=train_df[features], y=train_df[CFG.target_col])):\n        print('-'*50)\n        print(f'{method} training fold {fold+1}')\n\n        x_train = train_df[features].iloc[train_index]\n        y_train = train_df[CFG.target_col].iloc[train_index]\n        x_valid = train_df[features].iloc[valid_index]\n        y_valid = train_df[CFG.target_col].iloc[valid_index]\n        if method == 'lightgbm':\n            model, valid_pred = lightgbm_training(x_train, y_train, x_valid, y_valid)\n            ## 2. feature importance\n            importance_df = pd.DataFrame(model.feature_importance(), index=features, columns=['importance']).reset_index()\n            importance_df.to_csv(CFG.MODEL_DATA_PATH / f'{method}_fold{fold + 1}_seed{CFG.SEED}_ver{CFG.VER}_importance.csv', index=False)\n        if method == 'xgboost':\n            model, valid_pred = xgboost_training(x_train, y_train, x_valid, y_valid)\n        if method == 'catboost':\n            model, valid_pred = catboost_training(x_train, y_train, x_valid, y_valid)\n\n        # Save best model\n        pickle.dump(model, open(CFG.MODEL_DATA_PATH / f'{method}_fold{fold + 1}_seed{CFG.SEED}_ver{CFG.VER}.pkl', 'wb'))\n        # Add to out of folds array\n        oof_predictions[valid_index] = valid_pred\n        oof_fold[valid_index] = fold + 1\n        del x_train, x_valid, y_train, y_valid, model, valid_pred\n        gc.collect()\n\n    # Compute out of folds metric\n    score = r2_score(train_df[CFG.target_col], oof_predictions, sample_weight=train_df['weight'])\n    print(f'{method} our out of folds CV {CFG.metric} is {score}')\n    # Create a dataframe to store out of folds predictions\n    oof_df = pd.DataFrame({CFG.target_col: train_df[CFG.target_col], f'{method}_prediction': oof_predictions, 'fold': oof_fold})\n    oof_df.to_csv(CFG.OOF_DATA_PATH / f'oof_{method}_seed{CFG.SEED}_ver{CFG.VER}.csv', index = False)","metadata":{"execution":{"iopub.status.busy":"2024-10-14T20:59:02.912182Z","iopub.execute_input":"2024-10-14T20:59:02.912722Z","iopub.status.idle":"2024-10-14T20:59:02.939721Z","shell.execute_reply.started":"2024-10-14T20:59:02.912664Z","shell.execute_reply":"2024-10-14T20:59:02.938311Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def read_train_data():\n    data_pl_df_list = []\n    for i in range(10):\n        data_pl_df_list.append(pl.read_parquet(CFG.DATA_PATH / f'train.parquet/partition_id={i}/part-0.parquet').sample(10000))\n    return pl.concat(data_pl_df_list)","metadata":{"execution":{"iopub.status.busy":"2024-10-14T20:59:02.941497Z","iopub.execute_input":"2024-10-14T20:59:02.942056Z","iopub.status.idle":"2024-10-14T20:59:02.957279Z","shell.execute_reply.started":"2024-10-14T20:59:02.941997Z","shell.execute_reply":"2024-10-14T20:59:02.955778Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"original_features = ['feature_' + str(x).zfill(2) for x in range(78+1)]","metadata":{"execution":{"iopub.status.busy":"2024-10-14T20:59:02.958914Z","iopub.execute_input":"2024-10-14T20:59:02.959361Z","iopub.status.idle":"2024-10-14T20:59:02.972948Z","shell.execute_reply.started":"2024-10-14T20:59:02.959313Z","shell.execute_reply":"2024-10-14T20:59:02.971738Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_pl_df = read_train_data()","metadata":{"execution":{"iopub.status.busy":"2024-10-14T20:59:02.974444Z","iopub.execute_input":"2024-10-14T20:59:02.974852Z","iopub.status.idle":"2024-10-14T20:59:21.759270Z","shell.execute_reply.started":"2024-10-14T20:59:02.974788Z","shell.execute_reply":"2024-10-14T20:59:21.757979Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for method in CFG.METHOD_LIST:\n    gradient_boosting_model_cv_training(method, train_pl_df.to_pandas(), original_features)","metadata":{"execution":{"iopub.status.busy":"2024-10-14T20:59:21.762375Z","iopub.execute_input":"2024-10-14T20:59:21.762789Z","iopub.status.idle":"2024-10-14T21:00:13.780526Z","shell.execute_reply.started":"2024-10-14T20:59:21.762746Z","shell.execute_reply":"2024-10-14T21:00:13.779200Z"},"trusted":true},"execution_count":null,"outputs":[]}]}