{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84493,"databundleVersionId":9871156,"sourceType":"competition"}],"dockerImageVersionId":30804,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-17T12:52:55.410033Z","iopub.execute_input":"2024-12-17T12:52:55.410440Z","iopub.status.idle":"2024-12-17T12:52:56.924277Z","shell.execute_reply.started":"2024-12-17T12:52:55.410405Z","shell.execute_reply":"2024-12-17T12:52:56.922095Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"##Topic: A data science and AI competition hosted by Jane Street, a trading firm.\n\n##Key Details:\n\n#The competition involves building a model using real-world financial data.\n#Challenges include modeling fat-tailed distributions, non-stationary time series, and market behavior shifts.\n#Dataset: Time series with 79 features and 9 responders, focusing on forecasting responder_6.\n#Evaluation: Scored using a weighted zero-mean R-squared score (R2).\n#Phases: Training phase with historical data and a forecasting phase with future data.\n#Constraints: Maximum RAM usage is 30GB.","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T12:52:56.927005Z","iopub.execute_input":"2024-12-17T12:52:56.927483Z","iopub.status.idle":"2024-12-17T12:52:56.933540Z","shell.execute_reply.started":"2024-12-17T12:52:56.927443Z","shell.execute_reply":"2024-12-17T12:52:56.932247Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"##Approach\n\n#Using LightGBM (known for memory efficiency) with time series features, memory optimizations, and proper validation.\n\n##Implementation:\n\n#Memory-efficient data loading\n#Feature engineering with time-based features\n#Advanced validation strategy\n#Memory optimization techniques\n#Model training with early stopping","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T12:52:56.934848Z","iopub.execute_input":"2024-12-17T12:52:56.935311Z","iopub.status.idle":"2024-12-17T12:52:56.948095Z","shell.execute_reply.started":"2024-12-17T12:52:56.935258Z","shell.execute_reply":"2024-12-17T12:52:56.946776Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Install LightGBM\n%pip install lightgbm==3.3.5\n\nprint(\"LightGBM installed successfully.\")\n# Install required packages\n%pip install pandas pyarrow dask","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T12:52:56.951179Z","iopub.execute_input":"2024-12-17T12:52:56.951592Z","iopub.status.idle":"2024-12-17T12:53:21.462822Z","shell.execute_reply.started":"2024-12-17T12:52:56.951543Z","shell.execute_reply":"2024-12-17T12:53:21.461328Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# First, let's check if the data files exist and their basic properties\nimport os\nprint(\"Checking for data files...\")\nprint(os.listdir('/kaggle/input/jane-street-real-time-market-data-forecasting'))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T12:53:21.464399Z","iopub.execute_input":"2024-12-17T12:53:21.464802Z","iopub.status.idle":"2024-12-17T12:53:21.472420Z","shell.execute_reply.started":"2024-12-17T12:53:21.464765Z","shell.execute_reply":"2024-12-17T12:53:21.471030Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Use Dask to load the large Parquet file in a memory-efficient way\nimport dask.dataframe as dd\n\n# Load the dataset using Dask\ntry:\n    dask_df = dd.read_parquet('/kaggle/input/jane-street-real-time-market-data-forecasting/train.parquet')\n    print(\"Dataset loaded successfully with Dask.\")\n    print(\"Dataset columns:\", dask_df.columns)\n    print(\"Dataset size:\", len(dask_df))\nexcept Exception as e:\n    print(\"Error loading dataset with Dask:\", str(e))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T12:53:21.474139Z","iopub.execute_input":"2024-12-17T12:53:21.474612Z","iopub.status.idle":"2024-12-17T12:53:23.299553Z","shell.execute_reply.started":"2024-12-17T12:53:21.474560Z","shell.execute_reply":"2024-12-17T12:53:23.298253Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Import required libraries\nimport dask.dataframe as dd\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\n# Set style for better visualization\nplt.style.use('seaborn')\n\nprint(\"Loading data with Dask...\")\n# Load the data\ndf = dd.read_parquet('/kaggle/input/jane-street-real-time-market-data-forecasting/train.parquet')\n\n# Basic dataset information\nprint(\"\\\nDataset Overview:\")\nprint(\"Number of rows:\", df.shape[0].compute())\nprint(\"Number of columns:\", df.shape[1])\nprint(\"\\\nColumn names:\")\nprint(df.columns)\n\n# Get data types and memory usage\nprint(\"\\\nData Types and Memory Usage:\")\nprint(df.dtypes)\n\n# Basic statistics for numerical columns\nprint(\"\\\nBasic Statistics:\")\nprint(df.describe().compute())\n\n# Check for missing values\nprint(\"\\\nMissing Values Count:\")\nmissing_values = df.isnull().sum().compute()\nprint(missing_values[missing_values > 0])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T12:53:23.300833Z","iopub.execute_input":"2024-12-17T12:53:23.301311Z","iopub.status.idle":"2024-12-17T12:57:01.594435Z","shell.execute_reply.started":"2024-12-17T12:53:23.301277Z","shell.execute_reply":"2024-12-17T12:57:01.592453Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"##This code:\n\n##Loads and samples large datasets.\n\n#Cleans and preprocesses data.\n#Trains a linear regression model to predict responder_6.\n#Evaluates the model using MSE and R².\n#Sets up a prediction function for deployment using the Kaggle inference server.","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T12:57:01.597110Z","iopub.execute_input":"2024-12-17T12:57:01.598549Z","iopub.status.idle":"2024-12-17T12:57:01.604820Z","shell.execute_reply.started":"2024-12-17T12:57:01.598468Z","shell.execute_reply":"2024-12-17T12:57:01.603540Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"###Import Libraries and Initialize Variables\n\nimport pandas as pd\n\n# Initialize a list to hold samples from each file\nsamples = []\n\n###Step 2: Load Data from Multiple Files\n\nfor i in range(10):\n    # Define the path for each partition file\n    file_path = f\"/kaggle/input/jane-street-real-time-market-data-forecasting/train.parquet/partition_id={i}/part-0.parquet\"\n    \n    # Read the data from the Parquet file into a DataFrame\n    chunk = pd.read_parquet(file_path)\n    \n    # Take a sample of 100,000 rows from the current DataFrame\n    sample_chunk = chunk.sample(n=100000, random_state=42)\n    \n    # Append the sampled data to the list\n    samples.append(sample_chunk)\n\n###Step 3: Combine the Samples into a Single DataFrame\n\n# Combine all sampled data into one DataFrame\nsample_df = pd.concat(samples, ignore_index=True)\n\n# Display the first rows of the combined DataFrame\nsample_df.head()\n\n###Step 4: Import Additional Libraries for Model Training\n\nimport numpy as np\nimport pandas as pd\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.linear_model import LinearRegression\nfrom sklearn.metrics import mean_squared_error, r2_score\n\n###Step 5: Prepare Features and Target Data\n\n# Extract columns that start with 'feature_' as input features\nfeatures = sample_df.filter(regex='^feature_')  \n\n# Extract columns that start with 'responder_' as target variables\nresponders = sample_df.filter(regex='^responder_')  \n\n# Focus only on the 'responder_6' column as the target variable\nX = features.values  # Input features\ny = responders['responder_6'].values  # Target variable\n\n###Step 6: Split the Data into Train and Test Sets\n\n# Split the data into training and test sets (80% train, 20% test)\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)\n\n###Step 7: Handle Missing and Infinite Values\n\n# Replace NaN values in the training set with the median value of each column\nmedianas = np.nanmedian(X_train, axis=0)\nX_train = np.where(np.isnan(X_train), medianas, X_train)\n\n# Replace NaN values in the test set with the median values calculated from the training set\nX_test = np.where(np.isnan(X_test), medianas, X_test)\n\n# Replace any infinite values with 0.0 in both train and test sets\nX_train = np.nan_to_num(X_train, nan=0.0, posinf=0.0, neginf=0.0)\nX_test = np.nan_to_num(X_test, nan=0.0, posinf=0.0, neginf=0.0)\n\n# Ensure target variables have no NaN or infinite values\ny_train = np.nan_to_num(y_train, nan=0.0, posinf=0.0, neginf=0.0)\ny_test = np.nan_to_num(y_test, nan=0.0, posinf=0.0, neginf=0.0)\n\n###Step 8: Train a Linear Regression Model\n\n# Create a Linear Regression model and train it using the training data\nmodel = LinearRegression()\nmodel.fit(X_train, y_train)\n\n###Step 9: Evaluate the Model\n\n# Use the trained model to predict values for the test set\ny_pred = model.predict(X_test)\n\n# Calculate Mean Squared Error (MSE) and R^2 Score\nmse = mean_squared_error(y_test, y_pred)\nr2 = r2_score(y_test, y_pred)\n\n# Display evaluation results\nprint(f\"Mean Squared Error (MSE): {mse:.4f}\")\nprint(f\"R^2 Score: {r2:.4f}\")\n\n###Step 10: Display Predictions\n\n# Compare actual and predicted values for the first 5 test samples\npredictions = pd.DataFrame({\n    'Actual': y_test[:5],\n    'Predicted': y_pred[:5]\n})\nprint(predictions)\n\n###Step 11: Prepare the Prediction Function for Inference\nimport os\nimport polars as pl\nimport kaggle_evaluation.jane_street_inference_server\n\n# Define the prediction function for the inference server\ndef predict(test: pl.DataFrame, lags: pl.DataFrame | None) -> pl.DataFrame | pd.DataFrame:\n    global lags_\n    if lags is not None:\n        lags_ = lags\n\n    # Select columns that start with 'feature_' for the model input\n    feature_columns = [col for col in test.columns if col.startswith(\"feature_\")]\n    features = test.select(feature_columns).to_numpy()\n    \n    # Preprocess the features (replace NaN and infinite values)\n    features = np.where(np.isnan(features), medianas, features)\n    features = np.nan_to_num(features, nan=0.0, posinf=0.0, neginf=0.0)\n\n    # Generate predictions for 'responder_6'\n    responder_6_predictions = model.predict(features)\n    \n    # Create a new Polars DataFrame with 'row_id' and predictions\n    predictions = test.select(\"row_id\").with_columns(\n        pl.Series(\"responder_6\", responder_6_predictions)\n    )\n\n    # Validation checks\n    if isinstance(predictions, pl.DataFrame):\n        assert predictions.columns == ['row_id', 'responder_6']\n    elif isinstance(predictions, pd.DataFrame):\n        assert (predictions.columns == ['row_id', 'responder_6']).all()\n    else:\n        raise TypeError('The predict function must return a DataFrame')\n    assert len(predictions) == len(test)\n\n    return predictions\n\n###Step 12: Run the Inference Server\n\ninference_server = kaggle_evaluation.jane_street_inference_server.JSInferenceServer(predict)\nif os.getenv('KAGGLE_IS_COMPETITION_RERUN'):\n    inference_server.serve()\nelse:\n    inference_server.run_local_gateway(\n        (\n            '/kaggle/input/jane-street-real-time-market-data-forecasting/test.parquet',\n            '/kaggle/input/jane-street-real-time-market-data-forecasting/lags.parquet',\n        )\n    )\n\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T12:57:01.606275Z","iopub.execute_input":"2024-12-17T12:57:01.606677Z","iopub.status.idle":"2024-12-17T12:57:43.589253Z","shell.execute_reply.started":"2024-12-17T12:57:01.606637Z","shell.execute_reply":"2024-12-17T12:57:43.587767Z"}},"outputs":[],"execution_count":null}]}