{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84493,"databundleVersionId":9871156,"sourceType":"competition"}],"dockerImageVersionId":30786,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\n# Initialize a list to hold samples from each file\nsamples = []\n# Load a sample from each file\nfor i in range(10):\n    file_path = f\"/kaggle/input/jane-street-real-time-market-data-forecasting/train.parquet/partition_id={i}/part-0.parquet\"\n    chunk = pd.read_parquet(file_path)\n    \n    # Take a sample of the data (adjust sample size as needed)\n    sample_chunk = chunk.sample(n=500000, random_state=42)  # For example, 100 rows\n    samples.append(sample_chunk)\n# Concatenate all samples into one DataFrame if needed\nsample_df = pd.concat(samples, ignore_index=True)","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-11-17T03:06:28.520071Z","iopub.execute_input":"2024-11-17T03:06:28.523012Z","iopub.status.idle":"2024-11-17T03:07:29.590557Z","shell.execute_reply.started":"2024-11-17T03:06:28.522909Z","shell.execute_reply":"2024-11-17T03:07:29.589060Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sample_df.head()","metadata":{"execution":{"iopub.status.busy":"2024-11-17T03:07:29.593140Z","iopub.execute_input":"2024-11-17T03:07:29.593577Z","iopub.status.idle":"2024-11-17T03:07:29.628716Z","shell.execute_reply.started":"2024-11-17T03:07:29.593508Z","shell.execute_reply":"2024-11-17T03:07:29.627518Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport tensorflow as tf\nfrom tensorflow.keras import layers, models\n# Separate features and responders\nfeatures = sample_df.filter(regex='^feature_')\nresponders = sample_df.filter(regex='^responder_')\n# Convert to numpy arrays for TensorFlow\nX = features.values  # Features for input\n#y = responders.values  # Responders for output\n# Assuming you have a DataFrame `y_train` with all responders\ny = responders[['responder_6']].values  # Keep only responder_6\nX = np.nan_to_num(X, nan=3.0, posinf=0.0, neginf=0.0)\ny = np.nan_to_num(y, nan=3.0, posinf=0.0, neginf=0.0)","metadata":{"execution":{"iopub.status.busy":"2024-11-17T03:07:29.630405Z","iopub.execute_input":"2024-11-17T03:07:29.630808Z","iopub.status.idle":"2024-11-17T03:07:54.087233Z","shell.execute_reply.started":"2024-11-17T03:07:29.630761Z","shell.execute_reply":"2024-11-17T03:07:54.085971Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"Is_keras = False","metadata":{"execution":{"iopub.status.busy":"2024-11-17T03:07:54.088904Z","iopub.execute_input":"2024-11-17T03:07:54.089665Z","iopub.status.idle":"2024-11-17T03:07:54.096420Z","shell.execute_reply.started":"2024-11-17T03:07:54.089594Z","shell.execute_reply":"2024-11-17T03:07:54.095133Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport xgboost as xgb\nfrom sklearn.model_selection import train_test_split\nX_train, X_val, y_train, y_val = train_test_split(X, y, test_size=0.2, random_state=42)","metadata":{"execution":{"iopub.status.busy":"2024-11-17T03:07:54.100424Z","iopub.execute_input":"2024-11-17T03:07:54.100834Z","iopub.status.idle":"2024-11-17T03:08:02.591682Z","shell.execute_reply.started":"2024-11-17T03:07:54.100793Z","shell.execute_reply":"2024-11-17T03:08:02.590338Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Define XGBoost regressor\nmodel_xgb = xgb.XGBRegressor(\n    objective='reg:squarederror',  # Regression objective\n    n_estimators=1000,             # Number of trees\n    #learning_rate=0.5,            # Learning rate\n    max_depth=8,                  # Maximum tree depth\n    min_child_weight=50, \n    colsample_bytree=0.8, \n    subsample=0.8, \n    eta=0.3,\n    random_state=42               # Reproducibility\n)\n# Train the model\nmodel_xgb.fit(X_train, y_train, eval_set=[(X_val, y_val)], eval_metric='rmse', early_stopping_rounds=10, verbose = True)","metadata":{"execution":{"iopub.status.busy":"2024-11-17T03:08:02.593212Z","iopub.execute_input":"2024-11-17T03:08:02.593943Z","iopub.status.idle":"2024-11-17T03:22:56.349056Z","shell.execute_reply.started":"2024-11-17T03:08:02.593896Z","shell.execute_reply":"2024-11-17T03:22:56.347170Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y_pred = model_xgb.predict(X_val)","metadata":{"execution":{"iopub.status.busy":"2024-11-17T03:22:56.351680Z","iopub.execute_input":"2024-11-17T03:22:56.352273Z","iopub.status.idle":"2024-11-17T03:23:09.726499Z","shell.execute_reply.started":"2024-11-17T03:22:56.352209Z","shell.execute_reply":"2024-11-17T03:23:09.720394Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.metrics import mean_squared_error, r2_score\nmse = mean_squared_error(y_val, y_pred, squared=False)\nr2 = r2_score(y_val, y_pred)\nprint(f\"RMSE: {mse}\")\nprint(f\"R²: {r2}\")","metadata":{"execution":{"iopub.status.busy":"2024-11-17T03:23:09.731790Z","iopub.execute_input":"2024-11-17T03:23:09.732921Z","iopub.status.idle":"2024-11-17T03:23:09.809973Z","shell.execute_reply.started":"2024-11-17T03:23:09.732852Z","shell.execute_reply":"2024-11-17T03:23:09.806332Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import joblib\n# Save the model\njoblib.dump(model_xgb, \"xgboost_sklearn.pkl\")","metadata":{"execution":{"iopub.status.busy":"2024-11-17T03:23:09.814552Z","iopub.execute_input":"2024-11-17T03:23:09.815990Z","iopub.status.idle":"2024-11-17T03:23:09.990807Z","shell.execute_reply.started":"2024-11-17T03:23:09.815825Z","shell.execute_reply":"2024-11-17T03:23:09.987815Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from keras.optimizers import RMSprop\nclass GCRMSprop(RMSprop):\n    def get_gradients(self, loss, params):\n        # We here just provide a modified get_gradients() function since we are\n        # trying to just compute the centralized gradients.\n        grads = []\n        gradients = super().get_gradients()\n        for grad in gradients:\n            grad_len = len(grad.shape)\n            if grad_len > 1:\n                axis = list(range(grad_len - 1))\n                grad -= ops.mean(grad, axis=axis, keep_dims=True)\n            grads.append(grad)\n        return grads\noptimizer = GCRMSprop(learning_rate=1e-4)","metadata":{"execution":{"iopub.status.busy":"2024-11-17T03:23:09.994704Z","iopub.execute_input":"2024-11-17T03:23:09.995880Z","iopub.status.idle":"2024-11-17T03:23:10.388520Z","shell.execute_reply.started":"2024-11-17T03:23:09.995729Z","shell.execute_reply":"2024-11-17T03:23:10.387287Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from tensorflow.keras.regularizers import l2\n# Define the number of input and output nodes\ninput_dim = X.shape[1]  # Number of features (79)\noutput_dim = y.shape[1]  # Number of responders (9)\n# Define the model\nmodel = models.Sequential([\n    layers.Input(shape=(input_dim,)), # Input layer\n    layers.LayerNormalization(),\n   # layers.BatchNormalization(),\n  #  layers.Dense(128, activation='relu'),\n  #  layers.Dropout(0.2),\n    layers.Dense(64, activation='linear'),  # Encoder\n    layers.Dense(32, activation='linear'),  # Bottleneck layer (compression)\n    layers.Dense(64, activation='linear'),  # Decoder\n#    layers.Dense(128, activation='relu'), \n #   layers.Dropout(0.2),\n    layers.Dense(output_dim, activation='linear', kernel_regularizer=l2(0.001))  # Output layer for responders\n])\nmodel.compile(optimizer=\"adam\", loss='mse')","metadata":{"execution":{"iopub.status.busy":"2024-11-17T03:23:10.391900Z","iopub.execute_input":"2024-11-17T03:23:10.393147Z","iopub.status.idle":"2024-11-17T03:23:10.649417Z","shell.execute_reply.started":"2024-11-17T03:23:10.392696Z","shell.execute_reply":"2024-11-17T03:23:10.647007Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from tensorflow.keras.callbacks import LearningRateScheduler\ndef step_decay(epoch):\n    initial_lr = 0.01\n    drop = 0.5\n    epochs_drop = 5\n    lr = initial_lr * (drop ** (epoch // epochs_drop))\n    return lr\nlr_scheduler = LearningRateScheduler(step_decay)","metadata":{"execution":{"iopub.status.busy":"2024-11-17T03:23:10.652642Z","iopub.execute_input":"2024-11-17T03:23:10.653615Z","iopub.status.idle":"2024-11-17T03:23:10.675311Z","shell.execute_reply.started":"2024-11-17T03:23:10.653551Z","shell.execute_reply":"2024-11-17T03:23:10.673073Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from tensorflow.keras.callbacks import ReduceLROnPlateau\nreduce_lr = ReduceLROnPlateau(monitor='val_loss', factor=0.5, patience=2, min_lr=1e-6)","metadata":{"execution":{"iopub.status.busy":"2024-11-17T03:23:10.679666Z","iopub.execute_input":"2024-11-17T03:23:10.681124Z","iopub.status.idle":"2024-11-17T03:23:10.692829Z","shell.execute_reply.started":"2024-11-17T03:23:10.680985Z","shell.execute_reply":"2024-11-17T03:23:10.690733Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from tensorflow.keras.callbacks import EarlyStopping\n# Define EarlyStopping\nearly_stopping = EarlyStopping(\n    monitor='val_loss',    # Monitor validation loss\n    patience=10,            # Number of epochs to wait for improvement\n    min_delta=0.00001,       # Minimum change to qualify as an improvement\n    restore_best_weights=True  # Restore weights from the best epoch\n)\n\nif Is_keras:\n    history = model.fit(\n        X, y,\n        epochs=50,\n        batch_size=32,\n        validation_split=0.2,\n        callbacks=[early_stopping,reduce_lr]\n    )","metadata":{"execution":{"iopub.status.busy":"2024-11-17T03:23:10.704299Z","iopub.execute_input":"2024-11-17T03:23:10.705163Z","iopub.status.idle":"2024-11-17T03:23:10.719335Z","shell.execute_reply.started":"2024-11-17T03:23:10.705048Z","shell.execute_reply":"2024-11-17T03:23:10.717159Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"if Is_keras:\n    model.save(\"/kaggle/working/model.keras\")","metadata":{"execution":{"iopub.status.busy":"2024-11-17T03:23:10.722276Z","iopub.execute_input":"2024-11-17T03:23:10.723142Z","iopub.status.idle":"2024-11-17T03:23:10.735957Z","shell.execute_reply.started":"2024-11-17T03:23:10.723036Z","shell.execute_reply":"2024-11-17T03:23:10.733545Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport polars as pl\nimport kaggle_evaluation.jane_street_inference_server","metadata":{"execution":{"iopub.status.busy":"2024-11-17T03:23:10.738917Z","iopub.execute_input":"2024-11-17T03:23:10.740072Z","iopub.status.idle":"2024-11-17T03:23:11.655280Z","shell.execute_reply.started":"2024-11-17T03:23:10.739903Z","shell.execute_reply":"2024-11-17T03:23:11.652494Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import polars as pl\nimport numpy as np\n# Assuming `model` is your trained model\n# Assuming features required by the model are named 'feature_00', 'feature_01', etc.\ndef predict(test: pl.DataFrame, lags: pl.DataFrame | None) -> pl.DataFrame | pd.DataFrame:\n    \"\"\"Make a prediction.\"\"\"\n    global lags_\n    if lags is not None:\n        lags_ = lags\n    # Extract the features for the model input\n    feature_columns = [col for col in test.columns if col.startswith(\"feature_\")]\n    features = test.select(feature_columns).to_numpy()  # Convert to numpy array for model input\n    features = np.nan_to_num(features, nan=0.0, posinf=0.0, neginf=0.0)\n    # Generate predictions using the model\n    #model_predictions = model.predict(features)\n    if Is_keras:\n        responder_6_predictions = model.predict(features)[:,0]\n    else:\n        responder_6_predictions = model_xgb.predict(features)\n   # print(responder_6_predictions)    \n    #responder_6_predictions = model_predictions[:, 6]  # Assuming responder_6 is at index 6\n    # Create a new Polars DataFrame with row_id and responder_6 predictions\n    predictions = test.select(\"row_id\").with_columns(\n        pl.Series(\"responder_6\", responder_6_predictions)\n    )\n    print(predictions)\n    # Ensure the output format and length requirements\n    if isinstance(predictions, pl.DataFrame):\n        assert predictions.columns == ['row_id', 'responder_6']\n    elif isinstance(predictions, pd.DataFrame):\n        assert (predictions.columns == ['row_id', 'responder_6']).all()\n    else:\n        raise TypeError('The predict function must return a DataFrame')\n    \n    assert len(predictions) == len(test)\n    return predictions","metadata":{"execution":{"iopub.status.busy":"2024-11-17T03:23:11.659052Z","iopub.execute_input":"2024-11-17T03:23:11.661157Z","iopub.status.idle":"2024-11-17T03:23:11.686459Z","shell.execute_reply.started":"2024-11-17T03:23:11.660840Z","shell.execute_reply":"2024-11-17T03:23:11.683128Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"inference_server = kaggle_evaluation.jane_street_inference_server.JSInferenceServer(predict)\nif os.getenv('KAGGLE_IS_COMPETITION_RERUN'):\n    inference_server.serve()\nelse:\n    inference_server.run_local_gateway(\n        (\n            '/kaggle/input/jane-street-real-time-market-data-forecasting/test.parquet',\n            '/kaggle/input/jane-street-real-time-market-data-forecasting/lags.parquet',\n        )\n    )","metadata":{"execution":{"iopub.status.busy":"2024-11-17T03:23:11.689300Z","iopub.execute_input":"2024-11-17T03:23:11.690099Z","iopub.status.idle":"2024-11-17T03:23:12.312716Z","shell.execute_reply.started":"2024-11-17T03:23:11.690026Z","shell.execute_reply":"2024-11-17T03:23:12.309522Z"},"trusted":true},"outputs":[],"execution_count":null}]}