{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84493,"databundleVersionId":9871156,"sourceType":"competition"},{"sourceId":143921,"sourceType":"modelInstanceVersion","modelInstanceId":121960,"modelId":145068},{"sourceId":144362,"sourceType":"modelInstanceVersion","isSourceIdPinned":true,"modelInstanceId":122352,"modelId":145434},{"sourceId":144445,"sourceType":"modelInstanceVersion","isSourceIdPinned":true,"modelInstanceId":122421,"modelId":145506},{"sourceId":145029,"sourceType":"modelInstanceVersion","isSourceIdPinned":true,"modelInstanceId":122933,"modelId":145997},{"sourceId":145938,"sourceType":"modelInstanceVersion","isSourceIdPinned":true,"modelInstanceId":123761,"modelId":146822},{"sourceId":146374,"sourceType":"modelInstanceVersion","isSourceIdPinned":true,"modelInstanceId":124151,"modelId":147210}],"dockerImageVersionId":30786,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Project Code Explanation\n\nIn this updated notebook, several key enhancements have been introduced to optimize both data processing and model training, alongside a reduction in certain features based on recent analysis.\n\n## Data Loading Optimization\n\n   I implemented a `load_data` function to directly load data from disk with column and date filtering options. This helps to reduce unnecessary steps in data loading and, by using Polars’ lazy evaluation feature, avoids excessive memory usage. This approach is designed to efficiently load data, especially when dealing with large datasets.\n\n## K-fold Model Training\n\n   The model now utilizes K-fold training to ensure robust evaluation and improved generalization. Each fold is trained independently, and the results are aggregated to provide a comprehensive assessment of the model's performance.\n\n## ModelGroup Class\n\n   A new `ModelGroup` class has been introduced to manage the training and prediction of multiple models efficiently. This class allows for easy storage and retrieval of models, ensuring consistent prediction methods across the model group.\n\n## Feature Reduction and Acknowledgments\n\nIn this version, certain features have been removed based on analysis revealing excessive missing values or ineffective data. Focusing on higher-quality features is expected to boost model performance and efficiency.\n\nSpecial thanks to *“[Features Engineering on JaneStreet](https://www.kaggle.com/code/malakafaqahmad/features-engineering-on-janestreet)”*, which provided valuable insights for this approach.\n","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport polars as pl\nimport pandas as pd\nimport lightgbm as lgb\n\nimport os\n\nimport kaggle_evaluation.jane_street_inference_server\n\nimport joblib","metadata":{"execution":{"iopub.status.busy":"2024-10-25T14:51:20.820757Z","iopub.execute_input":"2024-10-25T14:51:20.821167Z","iopub.status.idle":"2024-10-25T14:51:25.249541Z","shell.execute_reply.started":"2024-10-25T14:51:20.821115Z","shell.execute_reply":"2024-10-25T14:51:25.248122Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"TARGET = 'responder_6'\nFEAT_COLS = [f\"feature_{i:02d}\" for i in range(79)]\n# FEAT_COLS = [\n# # 'weight',\n# # 'feature_00', 'feature_01', 'feature_02', 'feature_03', 'feature_04',\n# 'feature_05', 'feature_06', 'feature_07', 'feature_08', 'feature_09', 'feature_10',\n# 'feature_11', 'feature_12', 'feature_13', 'feature_14', 'feature_15', 'feature_16',\n# 'feature_17', 'feature_18', 'feature_19', 'feature_20', \n# # 'feature_21', \n# 'feature_22',\n# 'feature_23', 'feature_24', 'feature_25', \n# # 'feature_26', 'feature_27', \n# 'feature_28','feature_29', 'feature_30', \n# # 'feature_31', \n# 'feature_32', 'feature_33', 'feature_34',\n# 'feature_35', 'feature_36', 'feature_37', 'feature_38', 'feature_39', 'feature_40',\n# 'feature_41', 'feature_42', 'feature_43', 'feature_44', 'feature_45', 'feature_46',\n# 'feature_47', 'feature_48', 'feature_49', 'feature_50', 'feature_51', 'feature_52',\n# 'feature_53', 'feature_54', 'feature_55', 'feature_56', 'feature_57', 'feature_58',\n# 'feature_59', 'feature_60', 'feature_61', 'feature_62', 'feature_63', 'feature_64',\n# 'feature_65', 'feature_66', 'feature_67', 'feature_68', 'feature_69', 'feature_70',\n# 'feature_71', 'feature_72', 'feature_73', 'feature_74', 'feature_75', 'feature_76',\n# 'feature_77', 'feature_78'\n# ]","metadata":{"execution":{"iopub.status.busy":"2024-10-25T14:51:25.251498Z","iopub.execute_input":"2024-10-25T14:51:25.252008Z","iopub.status.idle":"2024-10-25T14:51:25.257847Z","shell.execute_reply.started":"2024-10-25T14:51:25.251971Z","shell.execute_reply":"2024-10-25T14:51:25.256669Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def load_data(date_id_range=None, time_id_range=None, columns=None, return_type='pl'):\n    \"\"\"\n    Load data from Parquet files with optional filtering on date_id, time_id, and selected columns.\n\n    Parameters:\n    - date_id_range (tuple, optional): Range of date_id to filter (start, end). Default is None, which means all dates.\n    - time_id_range (tuple, optional): Range of time_id to filter (start, end). Default is None, which means all times.\n    - columns (list, optional): List of columns to load. Default is None, which means all columns.\n    - return_type (str, optional): Type of data to return ('pl' for Polars DataFrame or 'pd' for Pandas DataFrame). Default is 'pl'.\n\n    Returns:\n    - pl.DataFrame or pd.DataFrame: The filtered data as a Polars or Pandas DataFrame.\n    \"\"\"\n    data_dir = '../input/jane-street-real-time-market-data-forecasting'\n    # Load data using Polars lazy loading (scan_parquet)\n    data = pl.scan_parquet(f\"{data_dir}/train.parquet\")\n\n    # Apply date_id filter if specified\n    if date_id_range is not None:\n        start_date, end_date = date_id_range\n        data = data.filter((pl.col(\"date_id\") >= start_date) & (pl.col(\"date_id\") <= end_date))\n\n    # Apply time_id filter if specified\n    if time_id_range is not None:\n        start_time, end_time = time_id_range\n        data = data.filter((pl.col(\"time_id\") >= start_time) & (pl.col(\"time_id\") <= end_time))\n\n    # Select specific columns if specified\n    if columns is not None:\n        data = data.select(columns)\n\n    # Collect the data to execute the lazy operations\n    if return_type == 'pd':\n        return data.collect().to_pandas()\n    else:\n        return data.collect()\n","metadata":{"execution":{"iopub.status.busy":"2024-10-25T14:51:25.259227Z","iopub.execute_input":"2024-10-25T14:51:25.259782Z","iopub.status.idle":"2024-10-25T14:51:25.275717Z","shell.execute_reply.started":"2024-10-25T14:51:25.259746Z","shell.execute_reply":"2024-10-25T14:51:25.274635Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def calculate_r2(y_true, y_pred, weights):\n    \"\"\"\n    Calculate the sample weighted zero-mean R-squared score (R2).\n\n    Parameters:\n    - y_true (pd.Series or np.array): Ground truth values.\n    - y_pred (pd.Series or np.array): Predicted values.\n    - weights (pd.Series or np.array): Sample weights.\n\n    Returns:\n    - float: R2 score.\n    \"\"\"\n    numerator = np.sum(weights * (y_true - y_pred) ** 2)\n    denominator = np.sum(weights * (y_true ** 2))\n    r2_score = 1 - (numerator / denominator)\n    return r2_score\n\n\n\ndef evaluate_model(model, test_data):\n    y_pred = model.predict(test_data[FEAT_COLS])\n    y_true = test_data[TARGET].to_numpy() \n    weights = test_data['weight'].to_numpy()  \n\n    # Calculate R2 score\n    r2_score = calculate_r2(y_true, y_pred, weights)\n    print(f\"Sample weighted zero-mean R-squared score (R2) on test data: {r2_score}\")\n","metadata":{"execution":{"iopub.status.busy":"2024-10-25T14:51:25.278634Z","iopub.execute_input":"2024-10-25T14:51:25.279078Z","iopub.status.idle":"2024-10-25T14:51:25.296755Z","shell.execute_reply.started":"2024-10-25T14:51:25.279030Z","shell.execute_reply":"2024-10-25T14:51:25.295596Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class ModelGroup:\n    def __init__(self):\n        self.models = []\n\n    def add_model(self, model):\n        \"\"\"Add a trained model to the group.\"\"\"\n        self.models.append(model)\n\n    def predict(self, test_data):\n        \"\"\"Make predictions using all models in the group and return the average.\"\"\"\n        preds = []\n        for model in self.models:\n            if isinstance(model, lgb.Booster):\n                pred = model.predict(test_data[FEAT_COLS])\n            elif isinstance(model, xgb.Booster):\n                pred = model.predict(xgb.DMatrix(test_data[FEAT_COLS]))\n            elif hasattr(model, 'predict'):  # For PyTorch or other models\n                pred = model.predict(test_data[FEAT_COLS])\n            else:\n                raise ValueError(\"Unsupported model type\")\n            \n            preds.append(pred)\n        \n        # Average the predictions from all models\n        avg_pred = np.mean(preds, axis=0)\n        return avg_pred\n    \n    @classmethod\n    def load(cls, file_path):\n        \"\"\"Load a model group from a file.\"\"\"\n        model_group = joblib.load(file_path)\n        return model_group\n\ndef train_lgb_kfold(total_days=1498, n_splits=5, save_models=False):\n    \n    # 将时间序列划分为 n_splits 个部分\n    fold_size = total_days // n_splits\n    folds = [(i * fold_size, min((i + 1) * fold_size - 1, total_days - 1)) for i in range(n_splits)]\n    \n    model_group = ModelGroup()\n\n    for fold_idx in range(n_splits):\n        valid_range = folds[fold_idx]\n        train_ranges = [folds[i] for i in range(n_splits) if i != fold_idx]\n\n        print(f\"Fold {fold_idx}: validation range {valid_range}, train parts: {train_ranges}\")\n\n        valid_data = load_data(date_id_range=valid_range, columns=[\"date_id\", \"weight\"] + FEAT_COLS + [TARGET], return_type='pl')\n        valid_weight = valid_data['weight'].to_pandas()\n\n        train_data = None\n        for train_range in train_ranges:\n            partial_train_data = load_data(date_id_range=train_range, columns=[\"date_id\", \"weight\"] + FEAT_COLS + [TARGET], return_type='pl')\n            if train_data is None:\n                train_data = partial_train_data\n            else:\n                train_data = train_data.vstack(partial_train_data)\n\n        train_weight = train_data['weight'].to_pandas()\n\n        train_ds = lgb.Dataset(train_data.select(FEAT_COLS+['weight']).to_pandas(), label=train_data[TARGET].to_pandas(), weight=train_weight)\n        valid_ds = lgb.Dataset(valid_data.select(FEAT_COLS+['weight']).to_pandas(), label=valid_data[TARGET].to_pandas(), weight=valid_weight, reference=train_ds)\n\n        LGB_PARAMS = {\n            'objective': 'regression_l2',\n            'metric': 'rmse',\n            'learning_rate': 0.05,\n            'num_leaves': 31\n            ,\n            'max_depth': -1,\n            'random_state': 42,\n            'device': 'gpu',\n        }\n\n        early_stopping_callback = lgb.early_stopping(100)\n        verbose_eval_callback = lgb.log_evaluation(period=50)\n\n        model = lgb.train(\n            LGB_PARAMS,\n            train_ds,\n            num_boost_round=1000,\n            valid_sets=[train_ds, valid_ds],\n            valid_names=['train', 'valid'],\n            callbacks=[early_stopping_callback, verbose_eval_callback],\n        )\n\n        # Evaluate model on validation data\n        y_valid_pred = model.predict(valid_data.select(FEAT_COLS+['weight']).to_pandas())\n        r2_score = calculate_r2(valid_data[TARGET].to_pandas(), y_valid_pred, valid_weight)\n        print(f\"Fold {fold_idx} validation R2 score: {r2_score}\")\n\n        # Add the model to the group\n        model_group.add_model(model)\n\n    # Optionally save the models (not required since they are in the class)\n    if save_models:\n        joblib.dump(model_group, \"lgb_model_group.pkl\")\n        print(\"Saved the model group to lgb_model_group.pkl\")\n    \n    return model_group","metadata":{"execution":{"iopub.status.busy":"2024-10-25T14:51:25.298676Z","iopub.execute_input":"2024-10-25T14:51:25.299047Z","iopub.status.idle":"2024-10-25T14:51:25.416367Z","shell.execute_reply.started":"2024-10-25T14:51:25.298985Z","shell.execute_reply":"2024-10-25T14:51:25.414996Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Model Train","metadata":{}},{"cell_type":"code","source":"# total_days = 1699\n# lgb_models = train_lgb_kfold(total_days=total_days,n_splits =5)","metadata":{"execution":{"iopub.status.busy":"2024-10-25T14:51:25.417985Z","iopub.execute_input":"2024-10-25T14:51:25.418363Z","iopub.status.idle":"2024-10-25T14:51:25.430020Z","shell.execute_reply.started":"2024-10-25T14:51:25.418327Z","shell.execute_reply":"2024-10-25T14:51:25.428832Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Model Load","metadata":{}},{"cell_type":"code","source":"lgb_model = ModelGroup.load(\"/kaggle/input/lgb_model_group_v3/other/default/1/lgb_model_group_v3.pkl\")","metadata":{"execution":{"iopub.status.busy":"2024-10-25T14:51:25.431337Z","iopub.execute_input":"2024-10-25T14:51:25.432407Z","iopub.status.idle":"2024-10-25T14:51:25.741137Z","shell.execute_reply.started":"2024-10-25T14:51:25.432370Z","shell.execute_reply":"2024-10-25T14:51:25.739727Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"lags_ : pl.DataFrame | None = None\n\n# Replace this function with your inference code.\n# You can return either a Pandas or Polars dataframe, though Polars is recommended.\n# Each batch of predictions (except the very first) must be returned within 10 minutes of the batch features being provided.\ndef predict(test: pl.DataFrame, lags: pl.DataFrame | None) -> pl.DataFrame | pd.DataFrame:\n    \"\"\"Make a prediction.\"\"\"\n    # All the responders from the previous day are passed in at time_id == 0. We save them in a global variable for access at every time_id.\n    # Use them as extra features, if you like.\n    global lags_\n    if lags is not None:\n        lags_ = lags\n\n    predictions = test.select(\n        'row_id',\n        pl.lit(0.0).alias('responder_6'),\n    )\n    \n    feat = test[FEAT_COLS+['weight']].to_pandas()\n\n    pred = lgb_model.predict(feat)\n\n    \n    predictions = predictions.with_columns(pl.Series('responder_6', pred.ravel()))\n    print(predictions)\n    # The predict function must return a DataFrame\n    assert isinstance(predictions, pl.DataFrame | pd.DataFrame)\n    # with columns 'row_id', 'responer_6'\n    assert list(predictions.columns) == ['row_id', 'responder_6']\n    # and as many rows as the test data.\n    assert len(predictions) == len(test)\n\n    return predictions","metadata":{"execution":{"iopub.status.busy":"2024-10-25T14:51:25.742655Z","iopub.execute_input":"2024-10-25T14:51:25.743009Z","iopub.status.idle":"2024-10-25T14:51:25.754409Z","shell.execute_reply.started":"2024-10-25T14:51:25.742970Z","shell.execute_reply":"2024-10-25T14:51:25.753190Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"inference_server = kaggle_evaluation.jane_street_inference_server.JSInferenceServer(predict)\n\nif os.getenv('KAGGLE_IS_COMPETITION_RERUN'):\n    inference_server.serve()\nelse:\n    inference_server.run_local_gateway(\n        (\n            '/kaggle/input/jane-street-real-time-market-data-forecasting/test.parquet',\n            '/kaggle/input/jane-street-real-time-market-data-forecasting/lags.parquet',\n        )\n    )","metadata":{"execution":{"iopub.status.busy":"2024-10-25T14:51:25.755706Z","iopub.execute_input":"2024-10-25T14:51:25.756041Z","iopub.status.idle":"2024-10-25T14:51:26.211791Z","shell.execute_reply.started":"2024-10-25T14:51:25.756007Z","shell.execute_reply":"2024-10-25T14:51:26.210666Z"},"trusted":true},"execution_count":null,"outputs":[]}]}