{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":84493,"databundleVersionId":9871156,"sourceType":"competition"}],"dockerImageVersionId":30787,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-30T15:38:34.902442Z","iopub.execute_input":"2024-12-30T15:38:34.902765Z","iopub.status.idle":"2024-12-30T15:38:35.267727Z","shell.execute_reply.started":"2024-12-30T15:38:34.902737Z","shell.execute_reply":"2024-12-30T15:38:35.266809Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\n# Initialize a list to hold samples from each file\nsamples = []\n# Load a sample from each file\nfor i in range(10):\n    file_path = f\"/kaggle/input/jane-street-real-time-market-data-forecasting/train.parquet/partition_id={i}/part-0.parquet\"\n    chunk = pd.read_parquet(file_path)\n    \n    # Take a sample of the data (adjust sample size as needed)\n    sample_chunk = chunk.sample(n=500000, random_state=42)  # For example, 100 rows\n    samples.append(sample_chunk)\n# Concatenate all samples into one DataFrame if needed\nsample_df = pd.concat(samples, ignore_index=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-30T15:38:35.269132Z","iopub.execute_input":"2024-12-30T15:38:35.26978Z","iopub.status.idle":"2024-12-30T15:40:00.36018Z","shell.execute_reply.started":"2024-12-30T15:38:35.269753Z","shell.execute_reply":"2024-12-30T15:40:00.359207Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(sample_df)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-30T15:40:00.361368Z","iopub.execute_input":"2024-12-30T15:40:00.36172Z","iopub.status.idle":"2024-12-30T15:40:00.665945Z","shell.execute_reply.started":"2024-12-30T15:40:00.361681Z","shell.execute_reply":"2024-12-30T15:40:00.664877Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport tensorflow as tf\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import Dense, Dropout\nfrom sklearn.metrics import mean_squared_error, r2_score\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import StandardScaler\nfrom tensorflow.keras.regularizers import l1\nfrom tensorflow.keras.layers import BatchNormalization\n\n# Define the weighted R² function\ndef weighted_r2(y_true, y_pred, weights):\n    y_true = np.ravel(y_true)\n    y_pred = np.ravel(y_pred)\n    weights = np.ravel(weights)\n    y_true_mean = np.average(y_true, weights=weights)\n    numerator = np.sum(weights * (y_true - y_pred) ** 2)\n    denominator = np.sum(weights * (y_true - y_true_mean) ** 2)\n    return 1 - (numerator / denominator)\n\n# Load your data (replace sample_df with your DataFrame)\n# For demonstration, assuming sample_df is already loaded\n# sample_df = ...\n\n# Separate features and responders\nfeatures = sample_df.filter(regex='^feature_')\nresponders = sample_df.filter(regex='^responder_')\nweights = sample_df['weight']\n\n# Convert to numpy arrays for TensorFlow\nX = features.values\ny = responders[['responder_6']].values  # Use only 'responder_6'\n\n# Handle NaN or infinite values\nX = np.nan_to_num(X, nan=0.0, posinf=0.0, neginf=0.0)\ny = np.nan_to_num(y, nan=0.0, posinf=0.0, neginf=0.0)\n\n# Standardize the features\nscaler = StandardScaler()\nX = scaler.fit_transform(X)\n\n# Split data into train, validation, and test1 sets\nX_train, X_temp, y_train, y_temp, weights_train, weights_temp = train_test_split(\n    X, y, weights, test_size=0.3, random_state=42\n)\nX_val, X_test1, y_val, y_test1, weights_val, weights_test1 = train_test_split(\n    X_temp, y_temp, weights_temp, test_size=0.5, random_state=42\n)\n\n# -------------------------------\n# Define the ANN Model\n# -------------------------------\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-30T19:01:28.039797Z","iopub.execute_input":"2024-12-30T19:01:28.040155Z","iopub.status.idle":"2024-12-30T19:07:01.606407Z","shell.execute_reply.started":"2024-12-30T19:01:28.040126Z","shell.execute_reply":"2024-12-30T19:07:01.605467Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Define the ANN Model with increased complexity and different optimizer\nmodel = Sequential([\n    Dense(256, activation='mish', input_shape=(X_train.shape[1],)),  # First hidden layer\n    BatchNormalization(),\n    Dropout(0.4),\n\n    Dense(128, activation='mish'),  # Second hidden layer\n    BatchNormalization(),\n    Dropout(0.4),\n\n    Dense(64, activation='mish'),  # Third hidden layer\n    BatchNormalization(),\n    Dropout(0.3),\n\n    Dense(32, activation='mish'),  # Fourth hidden layer\n    BatchNormalization(),\n    Dropout(0.2),\n\n    Dense(1)  # Output layer\n])\n\n# Compile the model with Adam optimizer and lower learning rate\nfrom tensorflow.keras.optimizers import Adam\noptimizer = Adam(learning_rate=0.001)\n\nmodel.compile(optimizer=optimizer, loss='mse', metrics=['mse'])\n\n# Learning rate scheduler\nlr_schedule = tf.keras.callbacks.ReduceLROnPlateau(\n    monitor='val_loss', factor=0.5, patience=3, verbose=1, min_lr=1e-5\n)\n\n# Early stopping\nearly_stopping = tf.keras.callbacks.EarlyStopping(\n    monitor='val_loss', patience=5, verbose=1, restore_best_weights=True\n)\n\n# Train the ANN Model\nhistory = model.fit(\n    X_train, y_train,\n    epochs=20,  # Increased epochs for better learning\n    batch_size=512,  # Smaller batch size\n    validation_data=(X_val, y_val),\n    callbacks=[lr_schedule, early_stopping],\n    verbose=1\n)\n\n# Evaluate the model\ny_train_pred = model.predict(X_train)\ny_val_pred = model.predict(X_val)\ny_test1_pred = model.predict(X_test1)\n\n# Calculate RMSE\ntrain_rmse = mean_squared_error(y_train, y_train_pred, squared=False)\nval_rmse = mean_squared_error(y_val, y_val_pred, squared=False)\ntest1_rmse = mean_squared_error(y_test1, y_test1_pred, squared=False)\n\n# Calculate R²\ntrain_r2 = r2_score(y_train, y_train_pred)\nval_r2 = r2_score(y_val, y_val_pred)\ntest1_r2 = r2_score(y_test1, y_test1_pred)\n\n# Calculate weighted R²\ntrain_weighted_r2 = weighted_r2(y_train, y_train_pred, weights_train)\nval_weighted_r2 = weighted_r2(y_val, y_val_pred, weights_val)\ntest1_weighted_r2 = weighted_r2(y_test1, y_test1_pred, weights_test1)\n\n# Print results\nprint(f\"Train RMSE: {train_rmse}\")\nprint(f\"Validation RMSE: {val_rmse}\")\nprint(f\"Test1 RMSE: {test1_rmse}\")\nprint(f\"Train R²: {train_r2}\")\nprint(f\"Validation R²: {val_r2}\")\nprint(f\"Test1 R²: {test1_r2}\")\nprint(f\"Train Weighted R²: {train_weighted_r2}\")\nprint(f\"Validation Weighted R²: {val_weighted_r2}\")\nprint(f\"Test1 Weighted R²: {test1_weighted_r2}\")\n\n# Save the improved model\nmodel.save(\"improved_ann_model.h5\")\nprint(\"Improved model saved as 'improved_ann_model.h5'\")\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#X = X.reshape((X.shape[0], 1, X.shape[1]))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-30T15:46:17.046139Z","iopub.status.idle":"2024-12-30T15:46:17.046452Z","shell.execute_reply.started":"2024-12-30T15:46:17.04631Z","shell.execute_reply":"2024-12-30T15:46:17.046324Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport numpy as np\nimport polars as pl\nimport tensorflow as tf\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import LSTM, Dense, Dropout\nfrom sklearn.metrics import mean_squared_error, r2_score\nfrom sklearn.model_selection import train_test_split\nimport kaggle_evaluation.jane_street_inference_server\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-30T19:00:42.670102Z","iopub.execute_input":"2024-12-30T19:00:42.67095Z","iopub.status.idle":"2024-12-30T19:00:42.675584Z","shell.execute_reply.started":"2024-12-30T19:00:42.670917Z","shell.execute_reply":"2024-12-30T19:00:42.674615Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import kaggle_evaluation.jane_street_inference_server\n\n# Global variable for lags\nlags_: pl.DataFrame | None = None\n\n# Inference Function\ndef predict(test: pl.DataFrame, lags: pl.DataFrame | None) -> pl.DataFrame:\n    \"\"\"\n    Perform inference and return predictions.\n\n    Args:\n        test (pl.DataFrame): The test dataset.\n        lags (pl.DataFrame | None): The lags data.\n\n    Returns:\n        pl.DataFrame: Predictions with 'row_id' and 'responder_6'.\n    \"\"\"\n    global lags_\n\n    # Handle lags\n    if lags is not None:\n        lags_ = lags\n\n    # Extract features for model input\n    feature_columns = [col for col in test.columns if col.startswith(\"feature_\")]\n    features = test.select(feature_columns).to_numpy()\n    features = np.nan_to_num(features, nan=0.0, posinf=0.0, neginf=0.0)\n    #features = features.reshape((features.shape[0], 1, features.shape[1]))\n    features = scaler.transform(features)\n\n    # Generate predictions\n    responder_6 = model.predict(features).flatten()\n\n    # Create a predictions dataframe\n    predictions = test.select(\"row_id\").with_columns(\n        pl.Series(\"responder_6\", responder_6)\n    )\n\n    print(predictions)  # Debugging print\n    # Ensure correct output format\n    assert predictions.columns == ['row_id', 'responder_6']\n    assert len(predictions) == len(test)\n\n    return predictions\n\n# Initialize the inference server\ninference_server = kaggle_evaluation.jane_street_inference_server.JSInferenceServer(predict)\n\nif os.getenv('KAGGLE_IS_COMPETITION_RERUN'):\n    inference_server.serve()\nelse:\n    # Debugging to ensure gateway runs locally\n    try:\n        inference_server.run_local_gateway(\n            (\n                '/kaggle/input/jane-street-real-time-market-data-forecasting/test.parquet',\n                '/kaggle/input/jane-street-real-time-market-data-forecasting/lags.parquet',\n            )\n        )\n    except Exception as e:\n        print(f\"Inference Server Error: {e}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-30T19:00:44.149772Z","iopub.execute_input":"2024-12-30T19:00:44.150471Z","iopub.status.idle":"2024-12-30T19:00:44.661738Z","shell.execute_reply.started":"2024-12-30T19:00:44.150436Z","shell.execute_reply":"2024-12-30T19:00:44.660801Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}