{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84493,"databundleVersionId":9871156,"sourceType":"competition"}],"dockerImageVersionId":30804,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport polars as pl\nimport xgboost as xgb\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import mean_squared_error\nimport gc\nfrom matplotlib import pyplot as plt\nfrom matplotlib.ticker import MaxNLocator, FormatStrFormatter, PercentFormatter\n\nimport os\nimport warnings\nimport kaggle_evaluation.jane_street_inference_server\nfrom sklearn.model_selection import GridSearchCV\nfrom sklearn.model_selection import KFold","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-10T15:08:08.992559Z","iopub.execute_input":"2024-12-10T15:08:08.993244Z","iopub.status.idle":"2024-12-10T15:08:12.793492Z","shell.execute_reply.started":"2024-12-10T15:08:08.993201Z","shell.execute_reply":"2024-12-10T15:08:12.79221Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"FEATURE_COLUMNS = [f'feature_{i:02}' for i in range(79)]\nTARGET_COLUMN = 'responder_6'\n\ndef process_chunk(partition_id, sample_frac=0.97):\n    file_path = f\"/kaggle/input/jane-street-real-time-market-data-forecasting/train.parquet/partition_id={partition_id}/part-0.parquet\"\n    data = pl.read_parquet(file_path).to_pandas()\n    data = data.sample(frac=sample_frac, random_state=2025)\n    \n    X = data[FEATURE_COLUMNS].fillna(3).values\n    y = data[TARGET_COLUMN].values\n    w = data['weight'].values\n\n    del data\n    gc.collect()\n\n    return X, y, w\n\nX_chunks, y_chunks, weight_chunks = [], [], []\n\nfor partition_id in range(5):  # Process only half of the data (5 out of 10 partitions)\n    print(f\"Processing partition {partition_id}...\")\n    X_chunk, y_chunk, weight_chunk = process_chunk(partition_id)\n    X_chunks.append(X_chunk)\n    y_chunks.append(y_chunk)\n    weight_chunks.append(weight_chunk)\n    \n    # Clear memory after processing each chunk\n    del X_chunk, y_chunk, weight_chunk\n    gc.collect()\n\n# Concatenate all chunks\nX = np.vstack(X_chunks)\ny = np.hstack(y_chunks)\nweights = np.hstack(weight_chunks)\n\ndel X_chunks, y_chunks, weight_chunks\ngc.collect()\n\nprint(f\"Final dataset shapes - X: {X.shape}, y: {y.shape}, weights: {weights.shape}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T12:59:08.595673Z","iopub.execute_input":"2024-12-08T12:59:08.596882Z","iopub.status.idle":"2024-12-08T13:01:10.464805Z","shell.execute_reply.started":"2024-12-08T12:59:08.596773Z","shell.execute_reply":"2024-12-08T13:01:10.463441Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"split = int(0.8 * len(X))\ntrain_X, val_X = X[:split], X[split:]\ntrain_y, val_y = y[:split], y[split:]\ntrain_weights, val_weights = weights[:split], weights[split:]\n\nprint(f\"Training data: {train_X.shape}, Validation data: {val_X.shape}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T13:01:34.950737Z","iopub.execute_input":"2024-12-08T13:01:34.951182Z","iopub.status.idle":"2024-12-08T13:01:34.957849Z","shell.execute_reply.started":"2024-12-08T13:01:34.951144Z","shell.execute_reply":"2024-12-08T13:01:34.956583Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"dtrain = xgb.DMatrix(train_X, label=train_y, weight=train_weights)\ndval = xgb.DMatrix(val_X, label=val_y, weight=val_weights)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T13:01:38.065037Z","iopub.execute_input":"2024-12-08T13:01:38.065422Z","iopub.status.idle":"2024-12-08T13:02:00.268811Z","shell.execute_reply.started":"2024-12-08T13:01:38.065389Z","shell.execute_reply":"2024-12-08T13:02:00.267276Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def weighted_r2(y_pred, dtrain):\n    y_true = dtrain.get_label()\n    weights = dtrain.get_weight()\n    weights /= np.sum(weights) \n\n    numerator = np.sum(weights * (y_true - y_pred) ** 2)\n    denominator = np.sum(weights * y_true ** 2)\n    score = 1 - (numerator / denominator)\n\n    return 'weighted_r2', score\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T13:02:14.119436Z","iopub.execute_input":"2024-12-08T13:02:14.119978Z","iopub.status.idle":"2024-12-08T13:02:14.127009Z","shell.execute_reply.started":"2024-12-08T13:02:14.11993Z","shell.execute_reply":"2024-12-08T13:02:14.125789Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"params = {\n    'objective': 'reg:squarederror',\n    'eval_metric': 'rmse', \n    'tree_method': 'hist', \n    'learning_rate': 0.05,\n    'max_depth': 6,\n    'subsample': 0.8,\n    'colsample_bytree': 0.8,\n    'lambda': 1.0, \n    'alpha': 0.0,   \n    'seed': 2025\n}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T13:02:15.940689Z","iopub.execute_input":"2024-12-08T13:02:15.941109Z","iopub.status.idle":"2024-12-08T13:02:15.946841Z","shell.execute_reply.started":"2024-12-08T13:02:15.941072Z","shell.execute_reply":"2024-12-08T13:02:15.945571Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Train with early stopping\nevals = [(dtrain, 'train'), (dval, 'eval')]\nmodel = xgb.train(\n    params,\n    dtrain,\n    num_boost_round=200,\n    evals=evals,\n    early_stopping_rounds=50,\n    feval=weighted_r2,\n    maximize=True,   \n    verbose_eval=50\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T13:02:18.008466Z","iopub.execute_input":"2024-12-08T13:02:18.008907Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_pred = model.predict(dtrain)\nval_pred = model.predict(dval)\n\n\ndef compute_weighted_r2(y_true, y_pred, weights):\n    weights /= np.sum(weights)\n    numerator = np.sum(weights * (y_true - y_pred) ** 2)\n    denominator = np.sum(weights * y_true ** 2)\n    return 1 - (numerator / denominator)\n\ntrain_r2 = compute_weighted_r2(train_y, train_pred, train_weights)\nval_r2 = compute_weighted_r2(val_y, val_pred, val_weights)\n\nprint(f\"Train Weighted R2: {train_r2:.4f}\")\nprint(f\"Validation Weighted R2: {val_r2:.4f}\")\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport xgboost as xgb\n\nmodel = xgb.Booster()\n\n\ndef predict(test, lags):\n   \n    if 'responder_6_lag' in lags.columns:\n        test['responder_6_lag'] = lags['responder_6_lag']\n\n\n    dtest = xgb.DMatrix(test.to_numpy())\n    predictions = model.predict(dtest)\n\n    return predictions","metadata":{"trusted":true,"scrolled":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport kaggle_evaluation.jane_street_inference_server as js_server\n\ninference_server = kaggle_evaluation.jane_street_inference_server.JSInferenceServer(predict)\n\nif os.getenv('KAGGLE_IS_COMPETITION_RERUN'):\n    inference_server.serve()\nelse:\n    inference_server.run_local_gateway(\n        (\n            '/kaggle/input/jane-street-real-time-market-data-forecasting/test.parquet',\n            '/kaggle/input/jane-street-real-time-market-data-forecasting/lags.parquet',\n        )\n    )","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T15:08:18.54842Z","iopub.execute_input":"2024-12-10T15:08:18.548888Z","iopub.status.idle":"2024-12-10T15:08:18.91023Z","shell.execute_reply.started":"2024-12-10T15:08:18.548851Z","shell.execute_reply":"2024-12-10T15:08:18.908603Z"}},"outputs":[],"execution_count":null}]}