{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":84493,"databundleVersionId":9871156,"sourceType":"competition"}],"dockerImageVersionId":30787,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# Importing libraries and inference server\nimport os\n\nimport polars as pl\nimport pandas as pd\nimport numpy as np\n\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.decomposition import PCA\n\nfrom lightgbm import LGBMRegressor\nfrom xgboost import XGBRegressor\nfrom catboost import CatBoostRegressor\nfrom sklearn.ensemble import VotingRegressor\n\nimport kaggle_evaluation.jane_street_inference_server","metadata":{"execution":{"iopub.status.busy":"2024-10-25T12:35:36.413356Z","iopub.execute_input":"2024-10-25T12:35:36.413747Z","iopub.status.idle":"2024-10-25T12:35:36.419880Z","shell.execute_reply.started":"2024-10-25T12:35:36.413712Z","shell.execute_reply":"2024-10-25T12:35:36.418876Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Exploratory Data Analysis","metadata":{}},{"cell_type":"code","source":"# Reading data\ndf = pl.scan_parquet(\"/kaggle/input/jane-street-real-time-market-data-forecasting/train.parquet\")","metadata":{"execution":{"iopub.status.busy":"2024-10-25T12:35:37.309248Z","iopub.execute_input":"2024-10-25T12:35:37.310102Z","iopub.status.idle":"2024-10-25T12:35:37.314609Z","shell.execute_reply.started":"2024-10-25T12:35:37.310058Z","shell.execute_reply":"2024-10-25T12:35:37.313668Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Sample using one partition\ndf = df.filter(pl.col(\"partition_id\") == 0).collect()  # This is only using the first partition for right now... .slice(0, 10000)\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2024-10-25T12:35:37.737406Z","iopub.execute_input":"2024-10-25T12:35:37.738359Z","iopub.status.idle":"2024-10-25T12:35:38.235222Z","shell.execute_reply.started":"2024-10-25T12:35:37.738302Z","shell.execute_reply":"2024-10-25T12:35:38.234437Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Preprocessing","metadata":{}},{"cell_type":"code","source":"df.describe()","metadata":{"execution":{"iopub.status.busy":"2024-10-25T12:35:38.870273Z","iopub.execute_input":"2024-10-25T12:35:38.870943Z","iopub.status.idle":"2024-10-25T12:35:44.561961Z","shell.execute_reply.started":"2024-10-25T12:35:38.870901Z","shell.execute_reply":"2024-10-25T12:35:44.561024Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"(df.null_count() * -1) + pl.Series(\"\", [df.shape[0]])","metadata":{"execution":{"iopub.status.busy":"2024-10-25T12:35:44.563746Z","iopub.execute_input":"2024-10-25T12:35:44.564050Z","iopub.status.idle":"2024-10-25T12:35:44.572600Z","shell.execute_reply.started":"2024-10-25T12:35:44.564017Z","shell.execute_reply":"2024-10-25T12:35:44.571630Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Train/test split\nfrom sklearn.model_selection import train_test_split\n\n# Dropping feature columns that were fully null\nfeature_drops = [\"feature_00\", \"feature_01\", \"feature_02\", \"feature_03\", \"feature_04\",\n               \"feature_21\", \"feature_26\", \"feature_27\", \"feature_31\"]\n\n# Creating list of all features\nall_feature_cols = [f\"feature_{i:02d}\" for i in range(79)]  # features 00 to 78\n\n# Selecting only the features I want\nfeatures = [\"symbol_id\"] + [col for col in all_feature_cols if col not in feature_drops]\n\n# Use with_row_index to add an index column for tracking rows\ndf = df.with_row_index(name=\"index\")  # Add row number index\n\n# Now, select the features and target (without 'index' in features)\nX = df.select(features)  # 'index' is excluded from the model's features\ny = df.select(\"responder_6\")\n\n# Add the 'index' column to X using hstack\nX_with_index = X.hstack(df.select(\"index\"))\n\n# Convert X_with_index to pandas for train-test split\nX_pd = X_with_index.to_pandas()\n\n# Perform train-test split\nX_train, X_test, y_train, y_test = train_test_split(X_pd, y.to_pandas(), random_state=42, test_size=0.20, shuffle=False)\n\n# Convert X_test back to Polars\nX_test = pl.from_pandas(X_test)\n\n# Extract the original test indices based on the 'index' column\ntest_indices = X_test[\"index\"].to_list()\n\n# Now, extract the weights for the test set using these indices\nweights_test = df.filter(pl.col(\"index\").is_in(test_indices)).select(\"weight\")\n\n# If needed, convert weights to a NumPy array\nweights_test_array = weights_test.to_numpy()\n","metadata":{"execution":{"iopub.status.busy":"2024-10-25T12:35:44.573805Z","iopub.execute_input":"2024-10-25T12:35:44.574164Z","iopub.status.idle":"2024-10-25T12:35:45.509829Z","shell.execute_reply.started":"2024-10-25T12:35:44.574120Z","shell.execute_reply":"2024-10-25T12:35:45.508783Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mean_imputer = SimpleImputer(strategy='mean')\nimputer = mean_imputer.fit(X)\nX = imputer.transform(X)\n\nscaler = StandardScaler()\ntrain_data_scaled = scaler.fit_transform(X)\npca = PCA().fit(train_data_scaled)\n\ncumulative_variance_ratio = np.cumsum(pca.explained_variance_ratio_)\n\nn_components_95 = np.argmax(cumulative_variance_ratio >= 0.95) + 1\nprint(n_components_95)","metadata":{"execution":{"iopub.status.busy":"2024-10-25T12:35:45.511753Z","iopub.execute_input":"2024-10-25T12:35:45.512088Z","iopub.status.idle":"2024-10-25T12:36:03.822518Z","shell.execute_reply.started":"2024-10-25T12:35:45.512051Z","shell.execute_reply":"2024-10-25T12:36:03.821484Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train.drop('index', axis = 1,inplace = True)\nX_test_pd = X_test.to_pandas()\nX_test_pd.drop('index', axis = 1,inplace = True)\nX_test = pl.from_pandas(X_test_pd)","metadata":{"execution":{"iopub.status.busy":"2024-10-25T12:48:46.178080Z","iopub.execute_input":"2024-10-25T12:48:46.178469Z","iopub.status.idle":"2024-10-25T12:48:46.311668Z","shell.execute_reply.started":"2024-10-25T12:48:46.178419Z","shell.execute_reply":"2024-10-25T12:48:46.310508Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Preprocessing pipeline\npreprocessing = Pipeline([\n    ('impute', SimpleImputer(strategy='mean')),\n    ('scaler', StandardScaler())\n])\n\nX_train_processed = preprocessing.fit_transform(X_train)\nX_test_processed = preprocessing.fit_transform(X_test)","metadata":{"execution":{"iopub.status.busy":"2024-10-25T12:51:07.137174Z","iopub.execute_input":"2024-10-25T12:51:07.137900Z","iopub.status.idle":"2024-10-25T12:51:11.265289Z","shell.execute_reply.started":"2024-10-25T12:51:07.137855Z","shell.execute_reply":"2024-10-25T12:51:11.264264Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Model","metadata":{}},{"cell_type":"code","source":"# Model creating using voting regressor\nmodel = VotingRegressor([('lgbm', LGBMRegressor(num_leaves = 127, n_estimators = 200, max_depth = 3, learning_rate = 0.05, device_type='gpu', verbose=-1)),\n                         ('xgb', XGBRegressor(n_estimators = 200, min_child_weight = 5, max_depth = 7, learning_rate = 0.01, device='cpu')),\n                         ('catr', CatBoostRegressor(learning_rate = 0.05, l2_leaf_reg = 5, iterations = 300, depth = 8, task_type='GPU', verbose=0))])","metadata":{"execution":{"iopub.status.busy":"2024-10-25T12:51:11.266993Z","iopub.execute_input":"2024-10-25T12:51:11.267313Z","iopub.status.idle":"2024-10-25T12:51:11.273297Z","shell.execute_reply.started":"2024-10-25T12:51:11.267278Z","shell.execute_reply":"2024-10-25T12:51:11.272239Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# from sklearn.model_selection import RandomizedSearchCV\n\n# # Define parameter grid\n# param_grid = {\n#     'lgbm__n_estimators': [100, 200, 300],\n#     'lgbm__max_depth': [3, 5, 7],\n#     'lgbm__learning_rate': [0.01, 0.05, 0.1],\n#     'lgbm__num_leaves': [31, 63, 127],\n    \n#     'xgb__n_estimators': [100, 200, 300],\n#     'xgb__max_depth': [3, 5, 7],\n#     'xgb__learning_rate': [0.01, 0.05, 0.1],\n#     'xgb__min_child_weight': [1, 3, 5],\n    \n#     'catr__iterations': [100, 200, 300],\n#     'catr__depth': [4, 6, 8],\n#     'catr__learning_rate': [0.01, 0.05, 0.1],\n#     'catr__l2_leaf_reg': [1, 3, 5]\n# }\n\n# # Perform grid search\n# grid_search = RandomizedSearchCV(estimator=model,\n#                            param_distributions=param_grid,\n#                            cv=5,\n#                            scoring='r2',\n#                            n_iter = 30)\n\n# grid_search.fit(X_train_processed, np.ravel(y_train))\n# from sklearn.metrics import r2_score\n# y_pred = grid_search.predict(X_test_processed)\n# print(r2_score(y_test, y_pred))\n\n# # Report best score and parameters\n# print(f\"Best score: {grid_search.best_score_:.3f}\")\n# print(f\"Best parameters: {grid_search.best_params_}\")\n\n# # Evaluate on test set\n# best_model = grid_search.best_estimator_\n# test_score = best_model.score(X_test, y_test)\n# print(f\"Test set score: {test_score:.3f}\")","metadata":{"execution":{"iopub.status.busy":"2024-10-25T12:51:11.274405Z","iopub.execute_input":"2024-10-25T12:51:11.274779Z","iopub.status.idle":"2024-10-25T12:51:11.283490Z","shell.execute_reply.started":"2024-10-25T12:51:11.274746Z","shell.execute_reply":"2024-10-25T12:51:11.282582Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Model fitting\nmodel.fit(X_train_processed, np.ravel(y_train))","metadata":{"execution":{"iopub.status.busy":"2024-10-25T12:51:11.285805Z","iopub.execute_input":"2024-10-25T12:51:11.286376Z","iopub.status.idle":"2024-10-25T12:52:46.860916Z","shell.execute_reply.started":"2024-10-25T12:51:11.286329Z","shell.execute_reply":"2024-10-25T12:52:46.859853Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import r2_score\ny_pred = model.predict(X_test_processed)\nprint(r2_score(y_test, y_pred)) # Currently 0.0177","metadata":{"execution":{"iopub.status.busy":"2024-10-25T12:52:46.862112Z","iopub.execute_input":"2024-10-25T12:52:46.863050Z","iopub.status.idle":"2024-10-25T12:53:02.363673Z","shell.execute_reply.started":"2024-10-25T12:52:46.863004Z","shell.execute_reply":"2024-10-25T12:53:02.362627Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Convert polars dataframe to pandas dataframe\nX_test_pd = X_test.to_pandas()  # Convert to pandas DataFrame\n\n# Predictions and scoring metric\ny_pred = model.predict(X_test_pd)\n\ndef weighted_zero_mean_r2(y_true, y_pred, weights):\n    y_true = np.array(y_true).flatten()\n    y_pred = np.array(y_pred).flatten()\n    weights = np.array(weights).flatten()\n\n    weighted_mean_true = np.average(y_true, weights=weights)\n    weighted_mean_pred = np.average(y_pred, weights=weights)\n\n    y_true_centered = y_true - weighted_mean_true\n    y_pred_centered = y_pred - weighted_mean_pred\n\n    numerator = np.sum(weights * y_true_centered * y_pred_centered) ** 2\n\n    denominator = np.sum(weights * y_true_centered ** 2) * np.sum(weights * y_pred_centered ** 2)\n\n    r_squared = numerator / denominator\n\n    return r_squared\n\n# Use the weight column from the pandas DataFrame\nr2 = weighted_zero_mean_r2(y_test, y_pred, weights_test)\nprint(f\"Weighted Zero-Mean R-squared Score: {r2:.4f}\")\n","metadata":{"execution":{"iopub.status.busy":"2024-10-25T12:53:02.365338Z","iopub.execute_input":"2024-10-25T12:53:02.365765Z","iopub.status.idle":"2024-10-25T12:53:05.283064Z","shell.execute_reply.started":"2024-10-25T12:53:02.365719Z","shell.execute_reply":"2024-10-25T12:53:05.281944Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Predictions","metadata":{}},{"cell_type":"code","source":"lags_ : pl.DataFrame | None = None\n\n# Replace this function with your inference code.\n# You can return either a Pandas or Polars dataframe, though Polars is recommended.\n# Each batch of predictions (except the very first) must be returned within 10 minutes of the batch features being provided.\ndef predict(test: pl.DataFrame, lags: pl.DataFrame | None) -> pl.DataFrame:\n    \"\"\"Make a prediction.\"\"\"\n    # All the responders from the previous day are passed in at time_id == 0. We save them in a global variable for access at every time_id.\n    # Use them as extra features, if you like.\n    global lags_\n    if lags is not None:\n        lags_ = lags\n\n    test_new = preprocessing.transform(test[features].to_pandas())\n    \n    # Create the predictions DataFrame\n    predictions = pl.DataFrame({\n        'row_id': test['row_id'],\n        'responder_6': model.predict(test_new)\n    })\n\n    # The predict function must return a DataFrame\n    assert isinstance(predictions, pl.DataFrame)\n    # with columns 'row_id', 'responer_6'\n    assert predictions.columns == ['row_id', 'responder_6']\n    # and as many rows as the test data.\n    assert len(predictions) == len(test)\n\n    return predictions","metadata":{"execution":{"iopub.status.busy":"2024-10-25T12:54:01.952807Z","iopub.execute_input":"2024-10-25T12:54:01.953567Z","iopub.status.idle":"2024-10-25T12:54:01.960796Z","shell.execute_reply.started":"2024-10-25T12:54:01.953519Z","shell.execute_reply":"2024-10-25T12:54:01.959824Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Keep this code so it can send the predictions\ninference_server = kaggle_evaluation.jane_street_inference_server.JSInferenceServer(predict)\n\nif os.getenv('KAGGLE_IS_COMPETITION_RERUN'):\n    inference_server.serve()\nelse:\n    inference_server.run_local_gateway(\n        (\n            '/kaggle/input/jane-street-real-time-market-data-forecasting/test.parquet',\n            '/kaggle/input/jane-street-real-time-market-data-forecasting/lags.parquet',\n        )\n    )","metadata":{"execution":{"iopub.status.busy":"2024-10-25T12:54:11.253029Z","iopub.execute_input":"2024-10-25T12:54:11.254135Z","iopub.status.idle":"2024-10-25T12:54:11.322785Z","shell.execute_reply.started":"2024-10-25T12:54:11.254073Z","shell.execute_reply":"2024-10-25T12:54:11.321739Z"},"trusted":true},"execution_count":null,"outputs":[]}]}