{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":84493,"databundleVersionId":9871156,"sourceType":"competition"},{"sourceId":187227,"sourceType":"modelInstanceVersion","modelInstanceId":159622,"modelId":181983}],"dockerImageVersionId":30805,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Start with all necessary imports\nimport numpy as np\nimport pandas as pd\nimport polars as pl\nimport pickle\nimport os\nimport xgboost as xgb\n\n# Load and configure the model\ntry:\n    with open(\"/kaggle/input/xgb_baseline/scikitlearn/default/1/xgb_model_lag.pkl\", 'rb') as file:\n        model = pickle.load(file)\n                    \n        # Print model parameters\n        print(\"\\nModel parameters:\")\n        print(f\"Tree method: {model.get_params().get('tree_method', 'Not set')}\")\n        print(f\"Device: {model.get_params().get('device', 'Not set')}\")\n        print(f\"Predictor: {model.get_params().get('predictor', 'Not set')}\")\n        # Configure model for CPU usage and stable prediction\n        model.set_params(\n            tree_method='hist',\n            device='cpu',\n            n_jobs=-1  # Use all available CPU cores\n        )\n            \n        # Print model parameters\n        print(\"\\nModel parameters:\")\n        print(f\"Tree method: {model.get_params().get('tree_method', 'Not set')}\")\n        print(f\"Device: {model.get_params().get('device', 'Not set')}\")\n        print(f\"Predictor: {model.get_params().get('predictor', 'Not set')}\")\nexcept Exception as e:\n    print(f\"Error loading model: {str(e)}\")\n    raise","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import xgboost as xgb\nimport sys\nimport torch  # Since Kaggle uses PyTorch's CUDA toolkit\n\ndef test_xgboost_gpu():\n    import numpy as np\n    \n    print(\"Testing XGBoost GPU capability:\")\n    \n    try:\n        # Create a small test dataset\n        X = np.random.rand(10, 3)\n        y = np.random.randint(0, 2, 10)\n        \n        # Try to create a DMatrix with GPU parameters\n        dtrain = xgb.DMatrix(X, label=y)\n        \n        # Try to train a small model with GPU\n        param = {\n            'tree_method': 'gpu_hist',\n            'device': 'cuda',\n            'predictor': 'gpu_predictor'\n        }\n        \n        # Run for just 1 round to test\n        bst = xgb.train(param, dtrain, num_boost_round=1)\n        print(\"Successfully trained a test model on GPU\")\n        \n        # Try prediction\n        pred = bst.predict(dtrain)\n        print(\"Successfully made predictions on GPU\")\n        \n        return True\n        \n    except Exception as e:\n        print(f\"GPU test failed: {str(e)}\")\n        return False\n\ngpu_available = test_xgboost_gpu()\nprint(f\"\\nCan use GPU: {gpu_available}\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def predict(test: pl.DataFrame, lags: pl.DataFrame | None) -> pl.DataFrame:\n    \"\"\"\n    Make predictions on test data with proper error handling and logging.\n    \n    Args:\n        test: Current batch of test data\n        lags: Previous day's data (provided when time_id == 0)\n    \n    Returns:\n        DataFrame with row_id and predictions for responder_6\n    \"\"\"\n    try:\n        # Store lags globally when provided\n        global lags_\n        if lags is not None:\n            lags_ = lags\n            print(f\"Received lags with columns: {lags_.columns}\")\n        \n        # Create working copy and log initial state\n        test_features = test.clone()\n        print(f\"Initial test features columns: {test_features.columns}\")\n        \n        # Add lagged features if available\n        if 'lags_' in globals():\n            lag_columns = [col for col in lags_.columns if 'responder' in col]\n            print(f\"Lag columns to be added: {lag_columns}\")  # Fixed string formatting\n            \n            test_features = test_features.join(\n                lags_.select(['symbol_id'] + lag_columns),\n                on='symbol_id',\n                how='left'\n            )\n            print(f\"Columns after adding lags: {test_features.columns}\")\n        \n        # Prepare features for prediction\n        columns_to_drop = ['row_id', 'is_scored', 'date_id', 'time_id']\n        print(f\"Initial columns to drop: {columns_to_drop}\")\n        \n        responder_cols = [col for col in test_features.columns \n                         if ('responder' in col and 'lag' not in col)]\n        columns_to_drop.extend(responder_cols)\n        print(f\"Columns to drop after adding responder columns: {columns_to_drop}\")\n        \n        # Only drop columns that exist\n        columns_to_drop = [col for col in columns_to_drop \n                          if col in test_features.columns]\n        print(f\"Final columns to drop: {columns_to_drop}\")\n        \n        # Convert to numpy array and make prediction\n        X_test = test_features.drop(columns_to_drop).to_numpy()\n        print(f\"Feature matrix shape: {X_test.shape}\")\n        \n        # Make prediction\n        y_pred = model.predict(X_test)\n        print(f\"Prediction shape: {y_pred.shape}\")\n        \n        # Create and validate predictions DataFrame\n        predictions = test.select('row_id').with_columns(pl.Series(\"responder_6\", y_pred))\n        \n        # Verify predictions format\n        print(predictions)\n        print(predictions.columns)\n        print(type(predictions))\n        \n        assert isinstance(predictions, pl.DataFrame), \"Predictions must be a Polars DataFrame\"\n        assert predictions.columns == ['row_id', 'responder_6'], \"Incorrect prediction columns\"\n        assert len(predictions) == len(test), \"Prediction length mismatch\"\n        \n        return predictions\n        \n    except Exception as e:\n        print(f\"Error during prediction: {str(e)}\")\n        raise\n\n\n\n# Set up the inference server with proper error handling\ntry:\n    import kaggle_evaluation.jane_street_inference_server\n    inference_server = kaggle_evaluation.jane_street_inference_server.JSInferenceServer(predict)\n    \n    if os.getenv('KAGGLE_IS_COMPETITION_RERUN'):\n        inference_server.serve()\n    else:\n        inference_server.run_local_gateway(\n            (\n                '/kaggle/input/jane-street-real-time-market-data-forecasting/test.parquet',\n                '/kaggle/input/jane-street-real-time-market-data-forecasting/lags.parquet',\n            )\n        )\nexcept Exception as e:\n    print(f\"Error in inference server setup: {str(e)}\")\n    raise","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}