{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":84493,"databundleVersionId":9871156,"sourceType":"competition"},{"sourceId":208004394,"sourceType":"kernelVersion"}],"dockerImageVersionId":30786,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Import","metadata":{}},{"cell_type":"code","source":"import os\nimport numpy as np\nimport polars as pl\nimport pandas as pd\nimport lightgbm as lgb\n\nimport warnings\nwarnings.filterwarnings(\"ignore\")\n\nimport kaggle_evaluation.jane_street_inference_server\nimport joblib","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Select and load the data","metadata":{}},{"cell_type":"code","source":"TARGET = 'responder_6'\nFEAT_COLS = [f\"feature_{i:02d}\" for i in range(79)]","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def load_data(date_id_range=None, time_id_range=None, columns=None, return_type='pl'):\n\n    data_dir = '../input/jane-street-real-time-market-data-forecasting'\n    # Load data using Polars lazy loading (scan_parquet)\n    data = pl.scan_parquet(f\"{data_dir}/train.parquet\")\n\n    # Apply date_id filter if specified\n    if date_id_range is not None:\n        start_date, end_date = date_id_range\n        data = data.filter((pl.col(\"date_id\") >= start_date) & (pl.col(\"date_id\") <= end_date))\n\n    # Apply time_id filter if specified\n    if time_id_range is not None:\n        start_time, end_time = time_id_range\n        data = data.filter((pl.col(\"time_id\") >= start_time) & (pl.col(\"time_id\") <= end_time))\n\n    # Select specific columns if specified\n    if columns is not None:\n        data = data.select(columns)\n\n    # Collect the data to execute the lazy operations\n    if return_type == 'pd':\n        return data.collect().to_pandas()\n    else:\n        return data.collect()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# def train_lgb_kfold_single(total_days=1498, n_splits=5, save_model=False): #1498\n#     # 将The time series is divided into n_splits\n#     fold_size = total_days // n_splits\n#     folds = [(i * fold_size, min((i + 1) * fold_size - 1, total_days - 1)) for i in range(n_splits)]\n\n#     cv_scores = []\n\n#     for fold_idx in range(n_splits):\n#         # The range of the current folded validation set\n#         valid_range = folds[fold_idx]\n#         # The remaining part is used as the training set\n#         train_ranges = [folds[i] for i in range(n_splits) if i != fold_idx]\n\n#         print(f\"Fold {fold_idx}: validation range {valid_range}, train parts: {train_ranges}\")\n\n#         # Load validation data\n#         valid_data = load_data(date_id_range=valid_range, columns=[\"date_id\", \"weight\"] + FEAT_COLS + [TARGET], return_type='pl')\n#         valid_weight = valid_data['weight'].to_pandas()\n\n#         # Load train data\n#         train_data = None\n#         for train_range in train_ranges:\n#             partial_train_data = load_data(date_id_range=train_range, columns=[\"date_id\", \"weight\"] + FEAT_COLS + [TARGET], return_type='pl')\n#             if train_data is None:\n#                 train_data = partial_train_data\n#             else:\n#                 train_data = train_data.vstack(partial_train_data)\n\n#         train_weight = train_data['weight'].to_pandas()\n\n#         # Default LightGBM parameters\n#         LGB_PARAMS = {\n#             'objective': 'regression_l2',\n#             'metric': 'rmse',\n#             'learning_rate': 0.05,\n#             'num_leaves': 31,\n#             'max_depth': -1,\n#             'random_state': 42,\n#             'device': 'gpu',\n#             'verbosity': -1,\n#         }\n\n#         # Build the LightGBM dataset\n#         train_ds = lgb.Dataset(train_data.select(FEAT_COLS + ['weight']).to_pandas(), label=train_data[TARGET].to_pandas(), weight=train_weight)\n#         valid_ds = lgb.Dataset(valid_data.select(FEAT_COLS + ['weight']).to_pandas(), label=valid_data[TARGET].to_pandas(), weight=valid_weight, reference=train_ds)\n\n#         # Callback function\n#         early_stopping_callback = lgb.early_stopping(100)\n#         verbose_eval_callback = lgb.log_evaluation(period=50)\n\n#         # Training the model\n#         model = lgb.train(\n#             LGB_PARAMS,\n#             train_ds,\n#             num_boost_round=1000,\n#             valid_sets=[train_ds, valid_ds],  \n#             valid_names=['train', 'valid'],\n#             callbacks=[early_stopping_callback, verbose_eval_callback],\n#         )\n\n#         # result\n#         y_valid_pred = model.predict(valid_data.select(FEAT_COLS + ['weight']).to_pandas())\n#         r2_score = calculate_r2(valid_data[TARGET].to_pandas(), y_valid_pred, valid_weight)\n#         print(f\"Fold {fold_idx} validation R2 score: {r2_score}\")\n\n#         # save each fold result\n#         cv_scores.append(r2_score)\n\n#     # final result\n#     print(f\"Cross-validation R2 scores: {cv_scores}\")\n#     print(f\"Mean R2 score: {np.mean(cv_scores)}, Std: {np.std(cv_scores)}\")\n\n#     return model, np.mean(cv_scores), np.std(cv_scores)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Model training and saving\n\n## Addressing <span style=\"color: red;\">\"Notebook Inference Server Error\"</span>\nFor your commit to be successful, the notebook execution time must be less than 900 seconds. If the execution time exceeds this limit, the commit is doomed to get a notebook inference server error even if the commit. So, we can divide the notes into two - one for training and one for submission.\n\n**Note here that the trained model needs to be saved, and the saved trained model is called at the time of submission for prediction, so as to shorten the submission time to accommodate the time limit.**","metadata":{}},{"cell_type":"code","source":"def train_lgb_kfold_single(total_days=1498, n_splits=5, save_model=True, save_path=\"models/\"):  \n    # Ensure save_path exists\n    if save_model and not os.path.exists(save_path):\n        os.makedirs(save_path)\n\n    # Split the time series into `n_splits` parts\n    fold_size = total_days // n_splits\n    folds = [(i * fold_size, min((i + 1) * fold_size - 1, total_days - 1)) for i in range(n_splits)]\n\n    cv_scores = []\n    model_group = []  # To save all models for each fold\n\n    for fold_idx in range(n_splits):\n        # The range of the current fold's validation set\n        valid_range = folds[fold_idx]\n        # The remaining parts are used as the training set\n        train_ranges = [folds[i] for i in range(n_splits) if i != fold_idx]\n\n        print(f\"Fold {fold_idx}: validation range {valid_range}, train parts: {train_ranges}\")\n\n        # Load validation data\n        valid_data = load_data(date_id_range=valid_range, columns=[\"date_id\", \"weight\"] + FEAT_COLS + [TARGET], return_type='pl')\n        valid_weight = valid_data['weight'].to_pandas()\n\n        # Load training data\n        train_data = None\n        for train_range in train_ranges:\n            partial_train_data = load_data(date_id_range=train_range, columns=[\"date_id\", \"weight\"] + FEAT_COLS + [TARGET], return_type='pl')\n            if train_data is None:\n                train_data = partial_train_data\n            else:\n                train_data = train_data.vstack(partial_train_data)\n\n        train_weight = train_data['weight'].to_pandas()\n\n        # Default LightGBM parameters\n        LGB_PARAMS = {\n            'objective': 'regression_l2',\n            'metric': 'rmse',\n            'learning_rate': 0.05,\n            'num_leaves': 31,\n            'max_depth': -1,\n            'random_state': 42,\n            'device': 'gpu',\n            'verbosity': -1,  # Suppress LightGBM logs\n        }\n\n        # Build the LightGBM dataset\n        train_ds = lgb.Dataset(train_data.select(FEAT_COLS + ['weight']).to_pandas(), label=train_data[TARGET].to_pandas(), weight=train_weight)\n        valid_ds = lgb.Dataset(valid_data.select(FEAT_COLS + ['weight']).to_pandas(), label=valid_data[TARGET].to_pandas(), weight=valid_weight, reference=train_ds)\n\n        # Callback functions\n        early_stopping_callback = lgb.early_stopping(100)\n        verbose_eval_callback = lgb.log_evaluation(period=50)\n\n        # Train the model\n        model = lgb.train(\n            LGB_PARAMS,\n            train_ds,\n            num_boost_round=1000,\n            valid_sets=[train_ds, valid_ds],  \n            valid_names=['train', 'valid'],\n            callbacks=[early_stopping_callback, verbose_eval_callback],\n        )\n\n        # Save the model for this fold\n        model_group.append(model)\n        # if save_model:\n        #     model_file = os.path.join(save_path, f\"lgb_model_fold_{fold_idx}.pkl\")\n        #     joblib.dump(model, model_file)\n        #     print(f\"Saved model for fold {fold_idx} to {model_file}\")\n\n        # Evaluate the model\n        y_valid_pred = model.predict(valid_data.select(FEAT_COLS + ['weight']).to_pandas())\n        r2_score = calculate_r2(valid_data[TARGET].to_pandas(), y_valid_pred, valid_weight)\n        print(f\"Fold {fold_idx} validation R2 score: {r2_score}\")\n\n        # Save each fold result\n        cv_scores.append(r2_score)\n\n    # Model fusion: The output of all models is averaged\n    print(f\"Total trained models: {len(model_group)}\")\n    final_model = model_group[0]  # The structure of the first model is used\n    print(\"Averaging models...\")\n    average_predictions = lambda data: average_models(model_group, data)\n\n    # Save the entire model group\n    if save_model:\n        joblib.dump(final_model, \"lgb_model.pkl\")\n        print(\"Saved the final merged model to lgb_model.pkl\")\n\n    # Print the cross-validation results\n    print(f\"Cross-validation R2 scores: {cv_scores}\")\n    print(f\"Mean R2 score: {np.mean(cv_scores)}, Std: {np.std(cv_scores)}\")\n\n    return final_model, np.mean(cv_scores), np.std(cv_scores)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def calculate_r2(y_true, y_pred, weights):\n    numerator = np.sum(weights * (y_true - y_pred) ** 2)\n    denominator = np.sum(weights * (y_true ** 2))\n    r2_score = 1 - (numerator / denominator)\n    return r2_score","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Train and load the model\n\n## *First version: training......*\n\nRemember to download the model file and upload it to the Kaggle dataset, annotate this part of the code in the second version, and import it from the input dataset.","metadata":{}},{"cell_type":"code","source":"# total_days = 1000 #1699\n# lgb_models, _, _ = train_lgb_kfold_single(total_days=total_days,n_splits =5)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"The model is trained from: [https://www.kaggle.com/code/nicolesy/lgbm-baseline-simple-model/notebook](http://)\n\nThe above model is saved to: [https://www.kaggle.com/datasets/nicolesy/lgbm-baseline-simple/settings](http://)\n\n## *Second version: loading......*","metadata":{}},{"cell_type":"code","source":"# Load the model from the saved file\n\ndef load_model(file_path=\"lgb_model.pkl\"): # lgb_model_group.pkl\n    model = joblib.load(file_path)\n    print(f\"Loaded the model group from {file_path}\")\n    return model","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"lgb_model = load_model(file_path=\"/kaggle/input/lgbm-baseline-simple-model/lgb_model.pkl\")\nprint(f\"Loaded model from the saved file.\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(f\"Type of lgb_model: {type(lgb_model)}\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Prediction","metadata":{}},{"cell_type":"code","source":"lags_ : pl.DataFrame | None = None\n\ndef predict(test: pl.DataFrame, lags: pl.DataFrame | None) -> pl.DataFrame | pd.DataFrame:\n\n    global lags_\n    if lags is not None:\n        lags_ = lags\n\n    predictions = test.select(\n        'row_id',\n        pl.lit(0.0).alias('responder_6'),\n    )\n    \n    feat = test[FEAT_COLS+['weight']].to_pandas()\n\n    pred = lgb_model.predict(feat)\n\n    \n    predictions = predictions.with_columns(pl.Series('responder_6', pred.ravel()))\n    print(predictions)\n\n    assert isinstance(predictions, pl.DataFrame | pd.DataFrame)\n\n    assert list(predictions.columns) == ['row_id', 'responder_6']\n\n    assert len(predictions) == len(test)\n\n    return predictions","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Submission","metadata":{}},{"cell_type":"code","source":"!ls /kaggle/input/jane-street-real-time-market-data-forecasting/test.parquet","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!ls /kaggle/input/jane-street-real-time-market-data-forecasting/lags.parquet","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"inference_server = kaggle_evaluation.jane_street_inference_server.JSInferenceServer(predict)\n\nif os.getenv('KAGGLE_IS_COMPETITION_RERUN'):\n    inference_server.serve()\nelse:\n    inference_server.run_local_gateway(\n        (\n            '/kaggle/input/jane-street-real-time-market-data-forecasting/test.parquet',\n            '/kaggle/input/jane-street-real-time-market-data-forecasting/lags.parquet',\n        )\n    )","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}