{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84493,"databundleVersionId":9871156,"sourceType":"competition"},{"sourceId":201238,"sourceType":"modelInstanceVersion","isSourceIdPinned":true,"modelInstanceId":171693,"modelId":194013},{"sourceId":201394,"sourceType":"modelInstanceVersion","isSourceIdPinned":true,"modelInstanceId":171821,"modelId":194141},{"sourceId":203414,"sourceType":"modelInstanceVersion","isSourceIdPinned":true,"modelInstanceId":173538,"modelId":194141},{"sourceId":203567,"sourceType":"modelInstanceVersion","isSourceIdPinned":true,"modelInstanceId":173664,"modelId":194141}],"dockerImageVersionId":30786,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport joblib \n\nimport pandas as pd\nimport polars as pl\nimport lightgbm as lgb\nimport xgboost as xgb\nimport catboost as cbt\nimport numpy as np \n\nfrom joblib import Parallel, delayed\n\nimport kaggle_evaluation.jane_street_inference_server\n\n# Define the path to the input data directory\n# If the local directory exists, use it; otherwise, use the Kaggle input directory\ninput_path = './jane-street-real-time-market-data-forecasting/' if os.path.exists('./jane-street-real-time-market-data-forecasting') else '/kaggle/input/jane-street-real-time-market-data-forecasting/'\n# input_path = '.'\n# Flag to determine if the script is in training mode or not\nTRAINING = False\n\n# Define the feature names based on the number of features (79 in this case)\nfeature_names = [f\"feature_{i:02d}\" for i in range(79)]\n\n# Number of validation dates to use\nnum_valid_dates = 100\n\n# Number of dates to skip from the beginning of the dataset\nskip_dates = 500\n\n# Number of folds for cross-validation\nN_fold = 4\n\n# If in training mode, load the training data\nif TRAINING:\n    # Load the training data from a Parquet file\n    df_raw = pl.read_parquet(f'{input_path}/train.parquet')\n    \n    # Filter the DataFrame to include only dates greater than or equal to skip_dates\n    df_raw = df_raw.filter(pl.col('date_id') >= skip_dates)\n\n    # Get unique dates from the DataFrame\n    dates = df_raw.select('date_id').unique().to_series()\n\n    # Define validation dates as the last `num_valid_dates` dates\n    valid_dates = dates[-num_valid_dates:]\n\n    # Define training dates as all dates except the last `num_valid_dates` dates\n    train_dates = dates[:-num_valid_dates]\n\n    # Display the last few rows of the DataFrame (for debugging purposes)\n    print(df_raw.tail())\n    \n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T10:43:11.716050Z","iopub.execute_input":"2024-12-19T10:43:11.716519Z","iopub.status.idle":"2024-12-19T10:43:11.727433Z","shell.execute_reply.started":"2024-12-19T10:43:11.716463Z","shell.execute_reply":"2024-12-19T10:43:11.725248Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!ls /kaggle/input/baseline_models_js/scikitlearn/default/1","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T10:43:11.730353Z","iopub.execute_input":"2024-12-19T10:43:11.730903Z","iopub.status.idle":"2024-12-19T10:43:12.927721Z","shell.execute_reply.started":"2024-12-19T10:43:11.730842Z","shell.execute_reply":"2024-12-19T10:43:12.926070Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Define the path to load pre-trained models (if not in training mode)\nmodel_path = '/kaggle/input/baseline_models_js/scikitlearn/resumed/1'\n# model_path = './baseline_models'\n# If in training mode, prepare validation data\n\n# Initialize a list to store trained models\nmodels = []\n\n# Function to train a model or load a pre-trained model\ndef train(df, model_dict, model_name='lgb'):\n    if TRAINING:\n        # Select dates for training based on the fold number\n        selected_dates = [date for ii, date in enumerate(train_dates) if ii % N_fold != i]\n        \n        # Get the model from the dictionary\n        model = model_dict[model_name]\n        \n        # Extract features, target, and weights for the selected training dates\n        X_train = df.filter(pl.col('date_id').is_in(selected_dates)).select(feature_names).to_numpy()\n        y_train = df.filter(pl.col('date_id').is_in(selected_dates)).select('responder_6').to_numpy().flatten()\n        w_train = df.filter(pl.col('date_id').is_in(selected_dates)).select('weight').to_numpy().flatten()\n        X_valid = df.filter(pl.col('date_id').is_in(valid_dates)).select(feature_names).to_numpy()\n        y_valid = df.filter(pl.col('date_id').is_in(valid_dates)).select('responder_6').to_numpy().flatten()\n        w_valid = df.filter(pl.col('date_id').is_in(valid_dates)).select('weight').to_numpy().flatten()\n        # Train the model based on the type (LightGBM, XGBoost, or CatBoost)\n        if model_name == 'lgb':\n            # Train LightGBM model with early stopping and evaluation logging\n            model.fit(X_train, y_train, w_train,  \n                      eval_metric=[r2_lgb],\n                      eval_set=[(X_valid, y_valid, w_valid)], \n                      callbacks=[\n                          lgb.early_stopping(100), \n                          lgb.log_evaluation(10)\n                      ])\n            \n        elif model_name == 'cbt':\n            # Prepare evaluation set for CatBoost\n            evalset = cbt.Pool(X_valid, y_valid, weight=w_valid)\n            \n            # Train CatBoost model with early stopping and verbose logging\n            model.fit(X_train, y_train, sample_weight=w_train, \n                      eval_set=[evalset], \n                      verbose=10, \n                      early_stopping_rounds=100)\n            \n        else:\n            # Train XGBoost model with early stopping and verbose logging\n            model.fit(X_train, y_train, sample_weight=w_train, \n                      eval_set=[(X_valid, y_valid)], \n                      sample_weight_eval_set=[w_valid], \n                      verbose=10, \n                      early_stopping_rounds=100)\n\n        # Append the trained model to the list\n        models.append(model)\n        \n        # Save the trained model to a file\n        joblib.dump(model, f'{model_path}/{model_name}_{i}.model')\n        \n        # Delete training data to free up memory\n        del X_train\n        del y_train\n        del w_train\n        \n        # Collect garbage to free up memory\n        import gc\n        gc.collect()\n        \n    else:\n        # If not in training mode, load the pre-trained model from the specified path\n        models.append(joblib.load(f'{model_path}/{model_name}_{i}.model'))\n        \n    return \n\n# Custom R2 metric for XGBoost\ndef r2_xgb(y_true, y_pred, sample_weight):\n    r2 = 1 - np.average((y_pred - y_true) ** 2, weights=sample_weight) / (np.average((y_true) ** 2, weights=sample_weight) + 1e-38)\n    return -r2\n\n# Custom R2 metric for LightGBM\ndef r2_lgb(y_true, y_pred, sample_weight):\n    r2 = 1 - np.average((y_pred - y_true) ** 2, weights=sample_weight) / (np.average((y_true) ** 2, weights=sample_weight) + 1e-38)\n    return 'r2', r2, True\n\n# Custom R2 metric for CatBoost\nclass r2_cbt(object):\n    def get_final_error(self, error, weight):\n        return 1 - error / (weight + 1e-38)\n\n    def is_max_optimal(self):\n        return True\n\n    def evaluate(self, approxes, target, weight):\n        assert len(approxes) == 1\n        assert len(target) == len(approxes[0])\n\n        approx = approxes[0]\n\n        error_sum = 0.0\n        weight_sum = 0.0\n\n        for i in range(len(approx)):\n            w = 1.0 if weight is None else weight[i]\n            weight_sum += w * (target[i] ** 2)\n            error_sum += w * ((approx[i] - target[i]) ** 2)\n\n        return error_sum, weight_sum\n\n# Dictionary to store different models with their configurations\nmodel_dict = {\n    'lgb': lgb.LGBMRegressor(n_estimators=500, device='gpu',gpu_device_id=0, gpu_platform_id=1, gpu_use_dp=True, objective='l2'),\n    'xgb': xgb.XGBRegressor(n_estimators=2000, learning_rate=0.1, max_depth=6, tree_method='hist', device=\"cuda\", objective='reg:squarederror', eval_metric=r2_xgb, disable_default_eval_metric=True),\n    'cbt': cbt.CatBoostRegressor(iterations=1000, learning_rate=0.05, task_type='GPU', loss_function='RMSE', eval_metric=r2_cbt()),\n}\n\n# Train models for each fold\nif TRAINING:\n    for k in range(5, 10):\n        df = df_raw.filter(pl.col(\"partition_id\") == k)\n        if TRAINING:\n        # Extract features, target, and weights for validation dates\n            \n            for i in range(N_fold):\n                train(df, model_dict, 'lgb')\n                train(df, model_dict, 'xgb')\n                train(df, model_dict, 'cbt')\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T10:43:12.930072Z","iopub.execute_input":"2024-12-19T10:43:12.930506Z","iopub.status.idle":"2024-12-19T10:43:12.957157Z","shell.execute_reply.started":"2024-12-19T10:43:12.930452Z","shell.execute_reply":"2024-12-19T10:43:12.955594Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"models = []\n# for model_name in [\n#                    #'lgb', \n#                    'xgb', \n#                    #'cbt'\n# ]:\n#     for k in range(5, 10):\n#         for i in range(N_fold):\n#             models.append(joblib.load(f'{model_path}/{model_name}_{k}_{i}.model'))\nmodels.append(joblib.load(f'{model_path}/xgb_{9}_{4}.model'))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T10:43:12.960176Z","iopub.execute_input":"2024-12-19T10:43:12.960478Z","iopub.status.idle":"2024-12-19T10:43:13.139120Z","shell.execute_reply.started":"2024-12-19T10:43:12.960447Z","shell.execute_reply":"2024-12-19T10:43:13.137839Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"lags_ : pl.DataFrame | None = None\n\n# Replace this function with your inference code.\n# You can return either a Pandas or Polars dataframe, though Polars is recommended.\n# Each batch of predictions (except the very first) must be returned within 10 minutes of the batch features being provided.\ndef predict(test: pl.DataFrame, lags: pl.DataFrame | None) -> pl.DataFrame | pd.DataFrame:\n    \"\"\"Make a prediction.\"\"\"\n    # All the responders from the previous day are passed in at time_id == 0. We save them in a global variable for access at every time_id.\n    # Use them as extra features, if you like.\n    \n    # print(test)\n    # print(lags)\n    \n    global lags_\n    if lags is not None:\n        lags_ = lags\n\n    predictions = test.select(\n        'row_id',\n        pl.lit(0.0).alias('responder_6'),\n    )\n    \n    feat = test[feature_names].to_numpy()\n    \n    pred = [model.predict(feat) for model in models]\n    pred = np.mean(pred, axis=0)\n    print(pred)\n    predictions = predictions.with_columns(pl.Series('responder_6', pred.ravel()))\n\n    # The predict function must return a DataFrame\n    assert isinstance(predictions, pl.DataFrame | pd.DataFrame)\n    # with columns 'row_id', 'responer_6'\n    assert list(predictions.columns) == ['row_id', 'responder_6']\n    # and as many rows as the test data.\n    assert len(predictions) == len(test)\n\n    return predictions\n    \ninference_server = kaggle_evaluation.jane_street_inference_server.JSInferenceServer(predict)\n\nif os.getenv('KAGGLE_IS_COMPETITION_RERUN'):\n    inference_server.serve()\nelse:\n    inference_server.run_local_gateway(\n        (\n            '/kaggle/input/jane-street-real-time-market-data-forecasting/test.parquet',\n            '/kaggle/input/jane-street-real-time-market-data-forecasting/lags.parquet',\n            # './test.parquet',\n            # './lags.parquet',\n        )\n    )\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T10:43:13.140269Z","iopub.execute_input":"2024-12-19T10:43:13.140670Z","iopub.status.idle":"2024-12-19T10:43:13.543304Z","shell.execute_reply.started":"2024-12-19T10:43:13.140620Z","shell.execute_reply":"2024-12-19T10:43:13.541728Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pl.read_parquet('./submission.parquet')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T10:43:13.544790Z","iopub.execute_input":"2024-12-19T10:43:13.545100Z","iopub.status.idle":"2024-12-19T10:43:13.560397Z","shell.execute_reply.started":"2024-12-19T10:43:13.545070Z","shell.execute_reply":"2024-12-19T10:43:13.559075Z"}},"outputs":[],"execution_count":null}]}