{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":84493,"databundleVersionId":9871156,"sourceType":"competition"}],"dockerImageVersionId":30823,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# Import libraries\nimport pandas as pd\nimport numpy as np\nimport os, gc\nfrom tqdm.auto import tqdm\nimport pickle\nfrom sklearn.metrics import r2_score\nfrom sklearn.model_selection import TimeSeriesSplit\nfrom xgboost import XGBRegressor\nimport optuna\nimport warnings\nimport polars as pl\n\nwarnings.filterwarnings('ignore')\npd.options.display.max_columns = None\n\n# Import Kaggle evaluation tools\nimport kaggle_evaluation.jane_street_inference_server","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-03T09:18:10.963301Z","iopub.execute_input":"2025-01-03T09:18:10.963571Z","iopub.status.idle":"2025-01-03T09:18:12.835199Z","shell.execute_reply.started":"2025-01-03T09:18:10.963527Z","shell.execute_reply":"2025-01-03T09:18:12.834506Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Configuration class\nclass CONFIG:\n    seed = 42\n    target_col = \"responder_6\"\n    feature_cols = [\n        \"symbol_id\", \"time_id\"\n    ] + [\n        f\"feature_{idx:02d}\" for idx in range(79)\n    ] + [\n        f\"responder_{idx}_lag_1\" for idx in range(3)\n    ]  # Only 3 lagged responder features\n\nprint(\"Step 1: Libraries imported and configuration set up.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-03T09:18:12.835893Z","iopub.execute_input":"2025-01-03T09:18:12.836235Z","iopub.status.idle":"2025-01-03T09:18:12.841482Z","shell.execute_reply.started":"2025-01-03T09:18:12.836214Z","shell.execute_reply":"2025-01-03T09:18:12.840452Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Function to reduce memory usage of data\ndef reduce_memory_usage(df):\n    for col in df.select_dtypes(include=['float', 'int']).columns:\n        col_min = df[col].min()\n        col_max = df[col].max()\n        if pd.api.types.is_float_dtype(df[col]):\n            df[col] = df[col].astype(np.float32)\n        elif pd.api.types.is_integer_dtype(df[col]):\n            if col_min >= 0:\n                if col_max < 255:\n                    df[col] = df[col].astype(np.uint8)\n                elif col_max < 65535:\n                    df[col] = df[col].astype(np.uint16)\n                else:\n                    df[col] = df[col].astype(np.uint32)\n            else:\n                if col_min > np.iinfo(np.int8).min and col_max < np.iinfo(np.int8).max:\n                    df[col] = df[col].astype(np.int8)\n                elif col_min > np.iinfo(np.int16).min and col_max < np.iinfo(np.int16).max:\n                    df[col] = df[col].astype(np.int16)\n                else:\n                    df[col] = df[col].astype(np.int32)\n    return df\n\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-03T09:18:12.842501Z","iopub.execute_input":"2025-01-03T09:18:12.842765Z","iopub.status.idle":"2025-01-03T09:18:12.859019Z","shell.execute_reply.started":"2025-01-03T09:18:12.842733Z","shell.execute_reply":"2025-01-03T09:18:12.858349Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Enhanced Feature Engineering\nlags_ = None  # Initialize lags_ globally\ndef add_features(df):\n    # Add lagged features for multiple columns\n    for lag in range(1, 3):  # Increase lag depth\n        for col in [\"feature_01\", \"feature_02\"]:\n            if col in df.columns:\n                df[f\"{col}_lag_{lag}\"] = df.groupby(\"symbol_id\")[col].shift(lag)\n\n    # Interaction features\n    if \"feature_01\" in df.columns and \"feature_02\" in df.columns:\n        df[\"feature_01_02_interaction\"] = df[\"feature_01\"] * df[\"feature_02\"]\n    if \"feature_03\" in df.columns and \"feature_04\" in df.columns:\n        df[\"feature_03_04_interaction\"] = df[\"feature_03\"] * df[\"feature_04\"]\n\n    # Rolling statistical features\n    df[\"feature_01_rolling_mean\"] = df.groupby(\"symbol_id\")[\"feature_01\"].rolling(3).mean().reset_index(level=0, drop=True)\n    df[\"feature_01_rolling_std\"] = df.groupby(\"symbol_id\")[\"feature_01\"].rolling(3).std().reset_index(level=0, drop=True)\n\n    # Lagged responder features (ensure these are created)\n    for idx in range(9):  # Use all responder lags\n        if f\"responder_{idx}\" in df.columns:\n            df[f\"responder_{idx}_lag_1\"] = df.groupby(\"symbol_id\")[f\"responder_{idx}\"].shift(1)\n\n    return df\n\n\ntrain_paths = [\n    f\"/kaggle/input/jane-street-real-time-market-data-forecasting/train.parquet/partition_id={i}\" for i in range(10)\n]\n\n# Load a subset of data for tuning, one partition at a time\ndef load_partition_for_tuning(partition_id):\n    print(f\"Loading data from partition_id={partition_id}...\")\n    path = f\"/kaggle/input/jane-street-real-time-market-data-forecasting/train.parquet/partition_id={partition_id}\"\n    chunk = pd.read_parquet(path)\n    chunk = reduce_memory_usage(chunk)  # Optimize memory\n    chunk = add_features(chunk)  # Add new features\n    return chunk\n\n# Use only one partition for tuning\nsubset_data = load_partition_for_tuning(0)\n\n# Prepare feature matrix, target, and weights\nX = subset_data[CONFIG.feature_cols].fillna(0)\ny = subset_data[CONFIG.target_col]\nw = subset_data[\"weight\"]\n\nprint(\"Subset Data Shape:\", subset_data.shape)\nprint(\"Feature Matrix Shape:\", X.shape)\nprint(\"Target Shape:\", y.shape)\nprint(\"Weights Shape:\", w.shape)\n\n# Clear memory\ndel subset_data\ngc.collect()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-03T09:18:13.259889Z","iopub.execute_input":"2025-01-03T09:18:13.260219Z","iopub.status.idle":"2025-01-03T09:18:19.71588Z","shell.execute_reply.started":"2025-01-03T09:18:13.260191Z","shell.execute_reply":"2025-01-03T09:18:19.715189Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import LSTM, Dense, Dropout\nfrom sklearn.preprocessing import StandardScaler\n\nscaler = StandardScaler()\nX_scaled = scaler.fit_transform(X)\n\nX_lstm = X_scaled.reshape(X_scaled.shape[0], 1, X_scaled.shape[1])\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-03T09:18:23.424996Z","iopub.execute_input":"2025-01-03T09:18:23.425342Z","iopub.status.idle":"2025-01-03T09:18:33.360623Z","shell.execute_reply.started":"2025-01-03T09:18:23.42531Z","shell.execute_reply":"2025-01-03T09:18:33.359882Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def create_lstm_model(input_shape):\n    model = Sequential()\n    model.add(LSTM(64, input_shape=input_shape, return_sequences=False))\n    model.add(Dropout(0.2))\n    model.add(Dense(1))\n    model.compile(optimizer='adam', loss='mean_squared_error')\n    return model","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-03T09:18:34.866219Z","iopub.execute_input":"2025-01-03T09:18:34.866762Z","iopub.status.idle":"2025-01-03T09:18:34.87115Z","shell.execute_reply.started":"2025-01-03T09:18:34.866737Z","shell.execute_reply":"2025-01-03T09:18:34.870022Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"lstm_model = create_lstm_model((X_lstm.shape[1], X_lstm.shape[2]))\nlstm_model.fit(X_lstm, y, epochs=10, batch_size=32)\n\n# Get LSTM predictions\nlstm_preds = lstm_model.predict(X_lstm)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-03T09:18:41.532868Z","iopub.execute_input":"2025-01-03T09:18:41.533193Z","iopub.status.idle":"2025-01-03T09:51:33.888996Z","shell.execute_reply.started":"2025-01-03T09:18:41.533161Z","shell.execute_reply":"2025-01-03T09:51:33.888011Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import xgboost as xgb\nxgb_model = xgb.XGBRegressor(n_estimators=100, learning_rate=0.05)\nxgb_model.fit(X, y)\n\n# Get XGBoost predictions\nxgb_preds = xgb_model.predict(X)\n\n# Stack both models' predictions\nstacked_preds = np.column_stack((lstm_preds, xgb_preds))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-03T09:52:10.928305Z","iopub.execute_input":"2025-01-03T09:52:10.928521Z","iopub.status.idle":"2025-01-03T09:52:47.921728Z","shell.execute_reply.started":"2025-01-03T09:52:10.928502Z","shell.execute_reply":"2025-01-03T09:52:47.920991Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.linear_model import LinearRegression\nfrom sklearn.metrics import r2_score\nmeta_model = LinearRegression()\nmeta_model.fit(stacked_preds, y)\n\n# Final predictions by combining both models\nfinal_preds = meta_model.predict(stacked_preds)\n\n# Evaluate the performance of the combined model\nprint(f\"R2 Score of the combined model: {r2_score(y, final_preds)}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-03T09:52:47.922648Z","iopub.execute_input":"2025-01-03T09:52:47.922939Z","iopub.status.idle":"2025-01-03T09:52:48.26112Z","shell.execute_reply.started":"2025-01-03T09:52:47.922914Z","shell.execute_reply":"2025-01-03T09:52:48.259683Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pickle\n\n# Save the trained model to a file\nwith open(\"meta_model.pkl\", \"wb\") as f:\n    pickle.dump(meta_model, f)\nprint(\"Model saved as 'meta_model.pkl'.\")\n","metadata":{"trusted":true,"execution":{"execution_failed":"2025-01-03T09:17:10.374Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Check if lags_ is None\nif lags_ is None:\n    print(\"lags_ is None. Please verify that it is being assigned correctly.\")\nelse:\n    print(\"lags_ is not None.\")","metadata":{"trusted":true,"execution":{"execution_failed":"2025-01-03T09:17:10.374Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def predict(test: pl.DataFrame, lags: pl.DataFrame | None) -> pl.DataFrame:\n    global lags_\n\n    # Handle lags and set up global variable\n    if lags is not None:\n        lags_ = lags\n\n    # Handle missing lags if they are not provided\n    if lags_ is None:\n        for idx in range(3):\n            test = test.with_columns(pl.lit(0.0).alias(f\"responder_{idx}_lag_1\"))\n\n    # Ensure lags_ is a Polars DataFrame\n    if isinstance(lags_, pd.DataFrame):  # Convert to Polars if it's a Pandas DataFrame\n        lags_ = pl.from_pandas(lags_)\n\n    # Merge lags into the test dataset if available\n    if lags_ is not None:\n        lags_last = lags_.select([\n            \"date_id\", \"symbol_id\", *[pl.col(f\"responder_{i}\").last().alias(f\"responder_{i}_last\") for i in range(9)]\n        ])\n\n        # Ensure the join keys have matching data types\n        test = test.with_columns([test[\"date_id\"].cast(pl.Int16), test[\"symbol_id\"].cast(pl.Int8)])\n        lags_last = lags_last.with_columns([lags_last[\"date_id\"].cast(pl.Int16), lags_last[\"symbol_id\"].cast(pl.Int8)])\n        test = test.join(lags_last, on=[\"date_id\", \"symbol_id\"], how=\"left\")\n\n    # Ensure all required feature columns are present in the test dataset\n    for col in CONFIG.feature_cols:\n        if col not in test.columns:\n            test = test.with_columns(pl.lit(0.0).alias(col))\n\n    # Fill missing values and prepare data for prediction\n    test_input = test[CONFIG.feature_cols].to_pandas().fillna(0)\n\n    # Convert to NumPy array and reshape for LSTM input\n    X_lstm_input = test_input.values.reshape(test_input.shape[0], 1, test_input.shape[1])\n\n    # Make predictions using the trained models\n    with open(\"meta_model.pkl\", \"rb\") as f:\n        final_model = pickle.load(f)\n\n    # Get predictions from the LSTM model\n    lstm_preds = lstm_model.predict(X_lstm_input)\n\n    # Get predictions from the XGBoost model\n    xgb_preds = xgb_model.predict(test_input)\n\n    # Stack both models' predictions\n    stacked_preds = np.column_stack((lstm_preds, xgb_preds))\n\n    # Final predictions using the meta model\n    final_preds = final_model.predict(stacked_preds)\n\n    # Prepare the output DataFrame\n    predictions_df = pl.DataFrame({\n        \"row_id\": test[\"row_id\"],\n        \"responder_6\": np.clip(final_preds, a_min=-5, a_max=5)  # Clip predictions to the range [-5, 5]\n    })\n\n    return predictions_df\n\n# Example test data (ensure features are consistent with your training data)\ntest_data = pl.DataFrame({\n    \"row_id\": [1, 2, 3],\n    \"date_id\": [1, 1, 1],\n    \"symbol_id\": [1, 2, 3],\n    \"feature_00\": [0.1, 0.2, 0.3],\n    \"feature_01\": [0.4, 0.5, 0.6],\n    # Add other features as required\n})\n\n# Get predictions\npredictions = predict(test_data, None)\nprint(predictions)\n","metadata":{"trusted":true,"execution":{"execution_failed":"2025-01-03T09:17:10.374Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"inference_server = kaggle_evaluation.jane_street_inference_server.JSInferenceServer(predict)\n\nif os.getenv('KAGGLE_IS_COMPETITION_RERUN'):\n    inference_server.serve()\nelse:\n    inference_server.run_local_gateway(\n        (\n            '/kaggle/input/jane-street-real-time-market-data-forecasting/test.parquet',\n            '/kaggle/input/jane-street-real-time-market-data-forecasting/lags.parquet',\n        )\n    )\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-03T07:43:05.476888Z","iopub.execute_input":"2025-01-03T07:43:05.47733Z","iopub.status.idle":"2025-01-03T07:43:05.724944Z","shell.execute_reply.started":"2025-01-03T07:43:05.477299Z","shell.execute_reply":"2025-01-03T07:43:05.723651Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Save the predictions to a submission file\nsubmission = predictions.to_pandas()  # Convert Polars DataFrame to Pandas DataFrame if needed\nsubmission.to_parquet(\"submission.parquet\", index=False)\nprint(\"Submission file saved as 'submission.parquet'.\")\n","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}