{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":84493,"databundleVersionId":9871156,"sourceType":"competition"},{"sourceId":187227,"sourceType":"modelInstanceVersion","modelInstanceId":159622,"modelId":181983}],"dockerImageVersionId":30804,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pickle\nwith open(\"/kaggle/input/xgb_baseline/scikitlearn/default/1/xgb_model_lag.pkl\", 'rb') as file:\n    model = pickle.load(file)\nmodel.set_params(tree_method='hist', device='cpu')\n\n#with open(\"/kaggle/working/xgb_model_lag.pkl\", 'wb') as file:\n#    model = pickle.dump(model, file)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#def predict(test: pl.DataFrame, lags: pl.DataFrame | None) -> pl.DataFrame:\n#    \"\"\"Make a prediction.\"\"\"\n#    # All the responders from the previous day are passed in at time_id == 0. We save them in a global variable for access at every time_id.\n#    # Use them as extra features, if you like.\n#    global lags_\n#    if lags is not None:\n#        lags_ = lags#\n\n#    \n#    X_test = test.drop(['row_id','is_scored',\"date_id\",\"time_id\"]).to_numpy()\n#    \n#    y_pred = model.predict(X_test)\n#        \n#    predictions = test.select('row_id').with_columns(pl.Series(\"responder_6\", y_pred))#\n\n#    # The predict function must return a DataFrame\n#    assert isinstance(predictions, pl.DataFrame)\n#    # with columns 'row_id', 'responer_6'\n#    assert predictions.columns == ['row_id', 'responder_6']\n#    # and as many rows as the test data.\n#    assert len(predictions) == len(test)#\n#\n#    return predictions","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import polars as pl\n\ndef predict(test: pl.DataFrame, lags: pl.DataFrame | None) -> pl.DataFrame:\n    \"\"\"\n    Make a prediction using the trained model.\n    \n    Args:\n        test: DataFrame containing the current test data\n        lags: DataFrame containing the lagged values (passed only when time_id == 0)\n    \n    Returns:\n        DataFrame with row_id and predictions for responder_6\n    \"\"\"\n    # Salviamo i lag quando vengono passati (time_id == 0)\n    global lags_\n    if lags is not None:\n        lags_ = lags\n        \n    # Creiamo una copia del DataFrame di test per non modificare l'originale\n    test_features = test.clone()\n    \n    # Se abbiamo i lag disponibili, aggiungiamo le feature laggate\n    if 'lags_' in globals():\n        # Assumiamo che i lag contengano le colonne responder_X_lag_1d\n        lag_columns = [col for col in lags_.columns if 'responder' in col]\n        \n        # Uniamo i lag al DataFrame di test usando symbol_id come chiave\n        test_features = test_features.join(\n            lags_.select(['symbol_id'] + lag_columns),\n            on='symbol_id',\n            how='left'\n        )\n    \n    # Rimuoviamo solo le colonne che sappiamo essere presenti nel DataFrame di test\n    # e che non vogliamo usare per la predizione\n    columns_to_drop = ['row_id', 'is_scored', 'date_id', 'time_id']\n    \n    # Aggiungiamo anche le colonne responder che non sono lag\n    columns_to_drop += [col for col in test_features.columns \n                       if ('responder' in col and 'lag' not in col)]\n    \n    # Rimuoviamo solo le colonne che effettivamente esistono\n    columns_to_drop = [col for col in columns_to_drop \n                      if col in test_features.columns]\n    \n    # Convertiamo in numpy array mantenendo solo le feature necessarie\n    X_test = test_features.drop(columns_to_drop).to_numpy()\n    \n    # Effettuiamo la predizione\n    y_pred = model.predict(X_test)\n    \n    # Creiamo il DataFrame di output nel formato richiesto\n    predictions = test.select('row_id').with_columns(pl.Series(\"responder_6\", y_pred))\n    \n    return predictions","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":" import kaggle_evaluation.jane_street_inference_server\n import os\n\n inference_server = kaggle_evaluation.jane_street_inference_server.JSInferenceServer(predict)\n if os.getenv('KAGGLE_IS_COMPETITION_RERUN'):\n     inference_server.serve()\n else:\n     inference_server.run_local_gateway(\n         (\n             '/kaggle/input/jane-street-real-time-market-data-forecasting/test.parquet',\n             '/kaggle/input/jane-street-real-time-market-data-forecasting/lags.parquet',\n         )\n     )","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}