{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":84493,"databundleVersionId":9871156,"sourceType":"competition"}],"dockerImageVersionId":30787,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-11-10T09:01:14.787710Z","iopub.execute_input":"2024-11-10T09:01:14.788341Z","iopub.status.idle":"2024-11-10T09:01:14.815798Z","shell.execute_reply.started":"2024-11-10T09:01:14.788302Z","shell.execute_reply":"2024-11-10T09:01:14.814918Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\n# Initialize a list to hold samples from each file\nsamples = []\n# Load a sample from each file\nfor i in range(10):\n    file_path = f\"/kaggle/input/jane-street-real-time-market-data-forecasting/train.parquet/partition_id={i}/part-0.parquet\"\n    chunk = pd.read_parquet(file_path)\n    \n    # Take a sample of the data (adjust sample size as needed)\n    sample_chunk = chunk.sample(n=100000, random_state=42)  # For example, 100 rows\n    samples.append(sample_chunk)\n# Concatenate all samples into one DataFrame if needed\nsample_df = pd.concat(samples, ignore_index=True)","metadata":{"execution":{"iopub.status.busy":"2024-11-10T09:01:19.492541Z","iopub.execute_input":"2024-11-10T09:01:19.492940Z","iopub.status.idle":"2024-11-10T09:02:18.210155Z","shell.execute_reply.started":"2024-11-10T09:01:19.492904Z","shell.execute_reply":"2024-11-10T09:02:18.209274Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sample_df.head()","metadata":{"execution":{"iopub.status.busy":"2024-11-10T09:02:18.212064Z","iopub.execute_input":"2024-11-10T09:02:18.212772Z","iopub.status.idle":"2024-11-10T09:02:18.244327Z","shell.execute_reply.started":"2024-11-10T09:02:18.212725Z","shell.execute_reply":"2024-11-10T09:02:18.243401Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport tensorflow as tf\nfrom tensorflow.keras import layers, models\nfrom tensorflow.keras.optimizers import Adam\n# Separate features and responders\nfeatures = sample_df.filter(regex='^feature_')\nresponders = sample_df.filter(regex='^responder_')\n# Convert to numpy arrays for TensorFlow\nX = features.values  # Features for input\ny = responders.values  # Responders for output\nX = np.nan_to_num(X, nan=0.0, posinf=0.0, neginf=0.0)\ny = np.nan_to_num(y, nan=0.0, posinf=0.0, neginf=0.0)","metadata":{"execution":{"iopub.status.busy":"2024-11-10T09:02:18.245363Z","iopub.execute_input":"2024-11-10T09:02:18.245673Z","iopub.status.idle":"2024-11-10T09:02:31.306776Z","shell.execute_reply.started":"2024-11-10T09:02:18.245641Z","shell.execute_reply":"2024-11-10T09:02:31.305948Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Define the number of input and output nodes\ninput_dim = X.shape[1]  # Number of features (79)\noutput_dim = y.shape[1]\n\nprint(\"Input dim \" , input_dim)\nprint(\"output_dim \" , output_dim)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-10T09:02:31.308779Z","iopub.execute_input":"2024-11-10T09:02:31.309484Z","iopub.status.idle":"2024-11-10T09:02:31.315879Z","shell.execute_reply.started":"2024-11-10T09:02:31.309436Z","shell.execute_reply":"2024-11-10T09:02:31.314753Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Define the number of input and output nodes\ninput_dim = X.shape[1]  # Number of features (79)\noutput_dim = y.shape[1]  # Number of responders (9)\n\n# Define the model\nmodel = models.Sequential([\n    layers.Input(shape=(input_dim,)),  # Input layer\n    \n    layers.Dense(512, activation='relu'),  \n    layers.BatchNormalization(),\n    layers.Dropout(0.3),\n    \n    layers.Dense(256, activation='relu'),\n    layers.BatchNormalization(),\n    layers.Dropout(0.3),\n    \n    layers.Dense(128, activation='relu'),\n    layers.BatchNormalization(),\n    layers.Dropout(0.3),\n    \n    layers.Dense(128, activation='relu'),  # Additional hidden layer\n    layers.BatchNormalization(),\n    layers.Dropout(0.2),\n    \n    layers.Dense(64, activation='relu'),\n    layers.BatchNormalization(),\n    layers.Dropout(0.2),\n    \n    layers.Dense(32, activation='relu'),  # Another additional hidden layer\n    layers.BatchNormalization(),\n    layers.Dropout(0.1),\n    \n    layers.Dense(output_dim, activation='linear')  # Output layer for responders\n])\n\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2024-11-10T09:06:02.194223Z","iopub.execute_input":"2024-11-10T09:06:02.194971Z","iopub.status.idle":"2024-11-10T09:06:03.052791Z","shell.execute_reply.started":"2024-11-10T09:06:02.194929Z","shell.execute_reply":"2024-11-10T09:06:03.051884Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model.compile(optimizer=Adam(learning_rate=1e-3), loss='mse', metrics=['mae'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-10T09:06:03.054423Z","iopub.execute_input":"2024-11-10T09:06:03.054755Z","iopub.status.idle":"2024-11-10T09:06:03.070202Z","shell.execute_reply.started":"2024-11-10T09:06:03.054720Z","shell.execute_reply":"2024-11-10T09:06:03.069063Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from tensorflow.keras.callbacks import EarlyStopping\n# Define EarlyStopping\nearly_stopping = EarlyStopping(\n    monitor='val_loss',   \n    patience=5,            \n    min_delta=0.001,       \n    restore_best_weights=True \n)\n\nhistory = model.fit(\n    X, y,\n    epochs=50,\n    batch_size=32,\n    validation_split=0.2,\n    callbacks=[early_stopping]\n)","metadata":{"execution":{"iopub.status.busy":"2024-11-10T09:06:03.472661Z","iopub.execute_input":"2024-11-10T09:06:03.473442Z","iopub.status.idle":"2024-11-10T09:16:00.681410Z","shell.execute_reply.started":"2024-11-10T09:06:03.473405Z","shell.execute_reply":"2024-11-10T09:16:00.680382Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Use the trained model to predict outputs based on the input features\npredictions = model.predict(X)\n# Display the predictions for the first 5 samples\nprint(predictions[:5])","metadata":{"execution":{"iopub.status.busy":"2024-11-10T09:17:55.920626Z","iopub.execute_input":"2024-11-10T09:17:55.921045Z","iopub.status.idle":"2024-11-10T09:18:56.647663Z","shell.execute_reply.started":"2024-11-10T09:17:55.921007Z","shell.execute_reply":"2024-11-10T09:18:56.646620Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport polars as pl\nimport kaggle_evaluation.jane_street_inference_server","metadata":{"execution":{"iopub.status.busy":"2024-11-10T09:18:56.649329Z","iopub.execute_input":"2024-11-10T09:18:56.649683Z","iopub.status.idle":"2024-11-10T09:18:56.654392Z","shell.execute_reply.started":"2024-11-10T09:18:56.649647Z","shell.execute_reply":"2024-11-10T09:18:56.653478Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import polars as pl\nimport numpy as np\n# Assuming `model` is your trained model\n# Assuming features required by the model are named 'feature_00', 'feature_01', etc.\ndef predict(test: pl.DataFrame, lags: pl.DataFrame | None) -> pl.DataFrame | pd.DataFrame:\n    \"\"\"Make a prediction.\"\"\"\n    global lags_\n    if lags is not None:\n        lags_ = lags\n    # Extract the features for the model input\n    feature_columns = [col for col in test.columns if col.startswith(\"feature_\")]\n    features = test.select(feature_columns).to_numpy()  # Convert to numpy array for model input\n    features = np.nan_to_num(features, nan=0.0, posinf=0.0, neginf=0.0)\n    # Generate predictions using the model\n    model_predictions = model.predict(features)\n    responder_6_predictions = model_predictions[:, 6]  # Assuming responder_6 is at index 6\n    # Create a new Polars DataFrame with row_id and responder_6 predictions\n    predictions = test.select(\"row_id\").with_columns(\n        pl.Series(\"responder_6\", responder_6_predictions)\n    )\n    # Ensure the output format and length requirements\n    if isinstance(predictions, pl.DataFrame):\n        assert predictions.columns == ['row_id', 'responder_6']\n    elif isinstance(predictions, pd.DataFrame):\n        assert (predictions.columns == ['row_id', 'responder_6']).all()\n    else:\n        raise TypeError('The predict function must return a DataFrame')\n    \n    assert len(predictions) == len(test)\n    return predictions","metadata":{"execution":{"iopub.status.busy":"2024-11-10T09:18:56.655354Z","iopub.execute_input":"2024-11-10T09:18:56.655675Z","iopub.status.idle":"2024-11-10T09:18:56.668603Z","shell.execute_reply.started":"2024-11-10T09:18:56.655642Z","shell.execute_reply":"2024-11-10T09:18:56.667731Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"inference_server = kaggle_evaluation.jane_street_inference_server.JSInferenceServer(predict)\nif os.getenv('KAGGLE_IS_COMPETITION_RERUN'):\n    inference_server.serve()\nelse:\n    inference_server.run_local_gateway(\n        (\n            '/kaggle/input/jane-street-real-time-market-data-forecasting/test.parquet',\n            '/kaggle/input/jane-street-real-time-market-data-forecasting/lags.parquet',\n        )\n    )","metadata":{"execution":{"iopub.status.busy":"2024-11-10T09:18:56.670234Z","iopub.execute_input":"2024-11-10T09:18:56.670531Z","iopub.status.idle":"2024-11-10T09:18:56.777379Z","shell.execute_reply.started":"2024-11-10T09:18:56.670499Z","shell.execute_reply":"2024-11-10T09:18:56.776617Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}