{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84493,"databundleVersionId":9871156,"sourceType":"competition"},{"sourceId":9726526,"sourceType":"datasetVersion","datasetId":5951701}],"dockerImageVersionId":30786,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-11-03T18:16:22.985740Z","iopub.execute_input":"2024-11-03T18:16:22.986169Z","iopub.status.idle":"2024-11-03T18:16:23.472799Z","shell.execute_reply.started":"2024-11-03T18:16:22.986126Z","shell.execute_reply":"2024-11-03T18:16:23.471444Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import polars as pl\nimport pandas as pd\nimport numpy as np\nfrom sklearn.linear_model import Ridge\nimport os\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport warnings\nwarnings.filterwarnings('ignore')\nimport matplotlib.pyplot as plt\nfrom statsmodels.tsa.seasonal import seasonal_decompose\nimport seaborn as sns\n\nimport kaggle_evaluation.jane_street_inference_server\n\nimport random\n\ndef seed_everything(seed):\n    np.random.seed(seed)\n    random.seed(seed)\n\nseed_everything(seed=2024)\n\ntarget = \"responder_6\"\nop_path = f\"/kaggle/working\"\nip_path = f\"/kaggle/input/janestreet2024-dataload-v1\"\nstate = 42\nmethod = \"LGBM1R\"","metadata":{"execution":{"iopub.status.busy":"2024-11-03T18:16:23.475068Z","iopub.execute_input":"2024-11-03T18:16:23.476107Z","iopub.status.idle":"2024-11-03T18:16:25.339052Z","shell.execute_reply.started":"2024-11-03T18:16:23.476051Z","shell.execute_reply":"2024-11-03T18:16:25.337954Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train=pl.read_parquet(\"/kaggle/input/jane-street-real-time-market-data-forecasting/train.parquet/partition_id=9/part-0.parquet\")\ntrain=train.to_pandas()\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2024-11-03T18:16:25.341876Z","iopub.execute_input":"2024-11-03T18:16:25.342553Z","iopub.status.idle":"2024-11-03T18:16:35.111895Z","shell.execute_reply.started":"2024-11-03T18:16:25.342498Z","shell.execute_reply":"2024-11-03T18:16:35.110566Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Generating the concated dataframe","metadata":{}},{"cell_type":"code","source":"def transform_dataframe(train: pd.DataFrame, idx: int) -> pd.DataFrame:\n    columns_to_keep = ['date_id', 'time_id', 'symbol_id', 'weight', 'responder_6']\n    dataframe = train[columns_to_keep]\n    dataframe = dataframe.rename(columns={\n        'date_id': f'data_id_{idx}',\n        'time_id': f'time_id_{idx}',\n        'symbol_id': f'symbol_id_{idx}',\n        'weight': f'weight_{idx}',\n        'responder_6': f'responder_6_{idx}'\n    })\n    return dataframe\n\n\n\ndef data_reader(path):\n    path = \"/kaggle/input/jane-street-real-time-market-data-forecasting/train.parquet/\" + path + \"/part-0.parquet\"\n    train=pl.read_parquet(path)\n    train=train.to_pandas()\n    return train","metadata":{"execution":{"iopub.status.busy":"2024-11-03T18:16:35.113263Z","iopub.execute_input":"2024-11-03T18:16:35.113637Z","iopub.status.idle":"2024-11-03T18:16:35.121423Z","shell.execute_reply.started":"2024-11-03T18:16:35.113598Z","shell.execute_reply":"2024-11-03T18:16:35.120276Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Read the generated dataframe","metadata":{}},{"cell_type":"code","source":"%%time\nresult = pd.read_csv('/kaggle/input/cancated-jane-street/cancated_dataframe.csv')\nresult.head()","metadata":{"execution":{"iopub.status.busy":"2024-11-03T18:16:35.124286Z","iopub.execute_input":"2024-11-03T18:16:35.124703Z","iopub.status.idle":"2024-11-03T18:17:43.304879Z","shell.execute_reply.started":"2024-11-03T18:16:35.124662Z","shell.execute_reply":"2024-11-03T18:17:43.303738Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(22, 6))\nrange_ = 3000\nresult['weight_0'][:range_].plot()\nresult['weight_1'][:range_].plot()\nresult['weight_2'][:range_].plot()\nresult['weight_3'][:range_].plot()\n\nplt.legend()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-11-03T18:17:43.306429Z","iopub.execute_input":"2024-11-03T18:17:43.306795Z","iopub.status.idle":"2024-11-03T18:17:44.055552Z","shell.execute_reply.started":"2024-11-03T18:17:43.306758Z","shell.execute_reply":"2024-11-03T18:17:44.054251Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(22, 6))\nresult['responder_6_0'][:500].plot()\nresult['responder_6_1'][:500].plot()\nresult['responder_6_2'][:500].plot()\n\n\n\nplt.legend()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-11-03T18:17:44.057340Z","iopub.execute_input":"2024-11-03T18:17:44.057807Z","iopub.status.idle":"2024-11-03T18:17:44.541975Z","shell.execute_reply.started":"2024-11-03T18:17:44.057754Z","shell.execute_reply":"2024-11-03T18:17:44.540853Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Filter columns that start with \"weight\" or \"responder\"\ncolumns_to_plot = [col for col in result.columns if col.startswith('weight') or col.startswith('responder')]\n\n# Plotting histograms and box plots for selected features\nfor col in columns_to_plot:\n    plt.figure(figsize=(12, 5))\n\n    # Histogram\n    plt.subplot(1, 2, 1)\n    plt.hist(result[col], bins=10, color='skyblue', edgecolor='black')\n    plt.title(f'Distribution of {col}')\n    plt.xlabel(col)\n    plt.ylabel('Frequency')\n\n    # Box plot\n    plt.subplot(1, 2, 2)\n    sns.boxplot(y=result[col], color='lightgreen')\n    plt.title(f'Box Plot of {col}')\n    plt.ylabel(col)\n\n    plt.tight_layout()\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2024-11-03T18:17:44.543381Z","iopub.execute_input":"2024-11-03T18:17:44.543755Z","iopub.status.idle":"2024-11-03T18:18:12.040686Z","shell.execute_reply.started":"2024-11-03T18:17:44.543716Z","shell.execute_reply":"2024-11-03T18:18:12.039484Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Creating the vertical dataframe","metadata":{}},{"cell_type":"code","source":"def extract_features_to_dataframe(result: pd.DataFrame) -> pd.DataFrame:\n    # Initialize lists for each feature\n    dates = []\n    times = []\n    symbols = []\n    weights = []\n    responders = []\n\n    # Iterate through the columns of the result DataFrame\n    for col in result.columns:\n        prefix = col.split('_')[0]\n        # Use extend to add values to the lists\n        if prefix == 'data':\n            dates.extend(result[col].values)\n        elif prefix == 'time':\n            times.extend(result[col].values)\n        elif prefix == 'symbol':\n            symbols.extend(result[col].values)\n        elif prefix == 'weight':\n            weights.extend(result[col].values)\n        elif prefix == 'responder':\n            responders.extend(result[col].values)\n\n    # Create a DataFrame from the lists\n    features_df = pd.DataFrame({\n        'dates': dates,\n        'times': times,\n        'symbols': symbols,\n        'weights': weights,\n        'responders': responders\n    })\n\n    return features_df\n\n# features_df = extract_features_to_dataframe(result)\n# print(features_df.describe())\n# features_df.head()","metadata":{"execution":{"iopub.status.busy":"2024-11-03T18:22:55.873791Z","iopub.execute_input":"2024-11-03T18:22:55.874562Z","iopub.status.idle":"2024-11-03T18:22:55.885331Z","shell.execute_reply.started":"2024-11-03T18:22:55.874468Z","shell.execute_reply":"2024-11-03T18:22:55.883862Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# plt.figure(figsize=(22, 6))\n# features_df['weights'].plot()\n# plt.show()","metadata":{"execution":{"iopub.status.busy":"2024-11-03T18:22:56.062955Z","iopub.execute_input":"2024-11-03T18:22:56.063448Z","iopub.status.idle":"2024-11-03T18:22:56.068670Z","shell.execute_reply.started":"2024-11-03T18:22:56.063402Z","shell.execute_reply":"2024-11-03T18:22:56.067451Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Visualization\n","metadata":{}},{"cell_type":"code","source":"# Set a Seaborn style for better aesthetics\nsns.set_style(\"whitegrid\")\n\n# Count the occurrences of each symbol\nsymbol_counts = train['symbol_id'].value_counts()\n\n# Create a bar plot\nplt.figure(figsize=(12, 6))\nbars = plt.bar(symbol_counts.index, symbol_counts.values, color=sns.color_palette(\"viridis\", len(symbol_counts)))\n\n# Add title and labels\nplt.title('Count of Each Symbol ID', fontsize=16)\nplt.xlabel('Symbol ID', fontsize=14)\nplt.ylabel('Count', fontsize=14)\n\n# Rotate x-ticks for better readability\nplt.xticks(rotation=45)\n\n# Optionally, add data labels on top of each bar\nfor bar in bars:\n    yval = bar.get_height()\n\n# Show the plot with tight layout\nplt.tight_layout()\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-11-03T18:22:57.270652Z","iopub.execute_input":"2024-11-03T18:22:57.271397Z","iopub.status.idle":"2024-11-03T18:22:58.124372Z","shell.execute_reply.started":"2024-11-03T18:22:57.271354Z","shell.execute_reply":"2024-11-03T18:22:58.123272Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 3D analysis","metadata":{}},{"cell_type":"code","source":"import plotly.graph_objects as go\n\ndef plot_3d_metrics_interactive(df, x_col, y_col, z_col, title):\n    \"\"\"\n    Create an interactive 3D plot for three specified columns in a DataFrame.\n\n    :param df: DataFrame containing the data.\n    :param x_col: Name of the column to be plotted on the x-axis.\n    :param y_col: Name of the column to be plotted on the y-axis.\n    :param z_col: Name of the column to be plotted on the z-axis.\n    :param title: Title of the plot.\n    \"\"\"\n    # Extract data for the specified columns\n    x_data = df[x_col]\n    y_data = df[y_col]\n    z_data = df[z_col]\n    \n    # Create a 3D scatter plot\n    fig = go.Figure(data=[go.Scatter3d(\n        x=x_data,\n        y=y_data,\n        z=z_data,\n        mode='markers',\n        marker=dict(\n            size=5,\n            color=z_data,  # Color based on the z-axis metric\n            colorscale='Viridis',\n            opacity=0.8,\n            colorbar=dict(title=z_col)\n        ),\n        text=[f'{x_col}: {x:.2f}, {y_col}: {y:.2f}, {z_col}: {z:.2f}' \n              for x, y, z in zip(x_data, y_data, z_data)],\n    )])\n\n    # Update layout\n    fig.update_layout(\n        scene=dict(\n            xaxis_title=x_col,\n            yaxis_title=y_col,\n            zaxis_title=z_col,\n        ),\n        title=title,\n        margin=dict(l=0, r=0, b=0, t=40)\n    )\n\n    return fig","metadata":{"execution":{"iopub.status.busy":"2024-11-03T18:22:59.400154Z","iopub.execute_input":"2024-11-03T18:22:59.401194Z","iopub.status.idle":"2024-11-03T18:22:59.428702Z","shell.execute_reply.started":"2024-11-03T18:22:59.401148Z","shell.execute_reply":"2024-11-03T18:22:59.427442Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_3d_metrics_interactive(result[:5000], 'time_id_0', 'weight_0', 'responder_6_0', title=\"3D Metrics Plot\")","metadata":{"execution":{"iopub.status.busy":"2024-11-03T18:22:59.642805Z","iopub.execute_input":"2024-11-03T18:22:59.643283Z","iopub.status.idle":"2024-11-03T18:23:00.396493Z","shell.execute_reply.started":"2024-11-03T18:22:59.643230Z","shell.execute_reply":"2024-11-03T18:23:00.395263Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_3d_metrics_interactive(result[:5000], 'data_id_0', 'symbol_id_0', 'responder_6_0', title=\"3D Metrics Plot\")","metadata":{"execution":{"iopub.status.busy":"2024-11-03T18:23:00.398465Z","iopub.execute_input":"2024-11-03T18:23:00.398841Z","iopub.status.idle":"2024-11-03T18:23:00.489519Z","shell.execute_reply.started":"2024-11-03T18:23:00.398800Z","shell.execute_reply":"2024-11-03T18:23:00.488518Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_3d_metrics_interactive(result[:5000], 'time_id_0', 'weight_0', 'symbol_id_0', title=\"3D Metrics Plot\")","metadata":{"execution":{"iopub.status.busy":"2024-11-03T18:23:00.491332Z","iopub.execute_input":"2024-11-03T18:23:00.492317Z","iopub.status.idle":"2024-11-03T18:23:00.579374Z","shell.execute_reply.started":"2024-11-03T18:23:00.492256Z","shell.execute_reply":"2024-11-03T18:23:00.578262Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_3d_metrics_interactive(result[:5000], 'time_id_0', 'data_id_0', 'weight_0', title=\"3D Metrics Plot\")","metadata":{"execution":{"iopub.status.busy":"2024-11-03T18:23:00.582036Z","iopub.execute_input":"2024-11-03T18:23:00.582532Z","iopub.status.idle":"2024-11-03T18:23:00.674654Z","shell.execute_reply.started":"2024-11-03T18:23:00.582482Z","shell.execute_reply":"2024-11-03T18:23:00.673545Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_3d_metrics_interactive(result[:5000], 'responder_6_0', 'responder_6_1', 'responder_6_2', title=\"3D Metrics Plot\")","metadata":{"execution":{"iopub.status.busy":"2024-11-03T18:23:00.675956Z","iopub.execute_input":"2024-11-03T18:23:00.676347Z","iopub.status.idle":"2024-11-03T18:23:00.765409Z","shell.execute_reply.started":"2024-11-03T18:23:00.676308Z","shell.execute_reply":"2024-11-03T18:23:00.764133Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Weight analysis","metadata":{}},{"cell_type":"markdown","source":"\nPortfolio weight represents the proportion of an individual asset's value relative to the total value of the portfolio. It indicates the percentage of the total investment allocated to a specific asset, helping to determine the exposure to that asset.\n\nFormula\nThe portfolio weight of an asset is calculated using the formula:\n\nPortfolio Weight of Asset = (Value of the Asset) / (Total Value of the Portfolio)\nValue of the Asset: The current market value of the individual asset.\nTotal Value of the Portfolio: The sum of the market values of all assets within the portfolio.\nExample\nIf the total portfolio value is $100,000 and one of the assets is worth $20,000, the portfolio weight of that asset would be:\n\n\nPortfolio Weight = 20,000 / 100,000 = 0.2 or 20%\n","metadata":{}},{"cell_type":"code","source":"df1 =train[train['symbol_id'] == 3]\ndf1.head()","metadata":{"execution":{"iopub.status.busy":"2024-11-03T18:23:02.232007Z","iopub.execute_input":"2024-11-03T18:23:02.232914Z","iopub.status.idle":"2024-11-03T18:23:02.409394Z","shell.execute_reply.started":"2024-11-03T18:23:02.232865Z","shell.execute_reply":"2024-11-03T18:23:02.408233Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(22, 6))\ndf1['weight'].plot()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-11-03T18:23:03.093278Z","iopub.execute_input":"2024-11-03T18:23:03.094142Z","iopub.status.idle":"2024-11-03T18:23:03.518223Z","shell.execute_reply.started":"2024-11-03T18:23:03.094096Z","shell.execute_reply":"2024-11-03T18:23:03.517030Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n# Set a Seaborn style for better aesthetics\nsns.set_style(\"whitegrid\")\n\n# Create the plot\nplt.figure(figsize=(16, 6))\nplt.plot(train['weight'][:500], color='royalblue', linewidth=2.5)\n\n# Add title and labels\nplt.title('Weight Over Time (First 500 Records)', fontsize=20)\nplt.xlabel('Index', fontsize=16)\nplt.ylabel('Weight', fontsize=16)\n\n# Add grid for better readability\nplt.grid(True)\n\n# Optionally, add markers to the line\nplt.scatter(range(500), train['weight'][:500], color='darkblue', s=10, alpha=0.5)\n\n# Show the plot with tight layout\nplt.tight_layout()\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-11-03T18:23:59.582066Z","iopub.execute_input":"2024-11-03T18:23:59.582498Z","iopub.status.idle":"2024-11-03T18:24:00.170427Z","shell.execute_reply.started":"2024-11-03T18:23:59.582457Z","shell.execute_reply":"2024-11-03T18:24:00.169261Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train[train['weight'] > 5][:2]","metadata":{"execution":{"iopub.status.busy":"2024-11-03T18:23:15.205067Z","iopub.execute_input":"2024-11-03T18:23:15.205752Z","iopub.status.idle":"2024-11-03T18:23:15.390314Z","shell.execute_reply.started":"2024-11-03T18:23:15.205681Z","shell.execute_reply":"2024-11-03T18:23:15.388990Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Weight Decomposition\n\nTime series decomposition is a technique used to break down a time series into several components that represent different aspects of the data. This method helps in understanding the underlying patterns within the time series and is commonly used in forecasting and anomaly detection.","metadata":{}},{"cell_type":"code","source":"decomposition = seasonal_decompose(train['weight'][:1000], model='additive', period=19)\n\n# Create a figure with a specific size\nfig, (ax1, ax2, ax3, ax4) = plt.subplots(4, 1, figsize=(18, 12))  # Change figsize as needed\n\n# Plot the decomposition components\ndecomposition.observed.plot(ax=ax1, legend=False, title='Observed')\ndecomposition.trend.plot(ax=ax2, legend=False, title='Trend')\ndecomposition.seasonal.plot(ax=ax3, legend=False, title='Seasonal')\ndecomposition.resid.plot(ax=ax4, legend=False, title='Residual')\n\n# Show the plot\nplt.tight_layout()\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-11-03T18:23:15.395658Z","iopub.execute_input":"2024-11-03T18:23:15.396684Z","iopub.status.idle":"2024-11-03T18:23:17.082031Z","shell.execute_reply.started":"2024-11-03T18:23:15.396637Z","shell.execute_reply":"2024-11-03T18:23:17.080738Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(14, 7))\n\n# Plot the first 1000 records only\nplt.scatter(train['time_id'][:1000], train['weight'][:1000], label='weight')\nplt.scatter(train['time_id'][:1000], train['responder_6'][:1000], label='responder_6')\n\nplt.legend()\nplt.title('Plot (First 1000 Records)')\nplt.xlabel('Date')\nplt.ylabel('Values')\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-11-03T18:23:17.084118Z","iopub.execute_input":"2024-11-03T18:23:17.084557Z","iopub.status.idle":"2024-11-03T18:23:17.598929Z","shell.execute_reply.started":"2024-11-03T18:23:17.084513Z","shell.execute_reply":"2024-11-03T18:23:17.597630Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Analysis Distribution of Features","metadata":{}},{"cell_type":"code","source":"sns.set_style(\"whitegrid\")\n\n# List of features to plot\nfeatures = train.columns[3:79]\n\n# Create the figure\nplt.figure(figsize=(12, 8))\n\n# Loop through each feature and plot the histogram\nfor feature in features[:6]:\n    sns.histplot(train[feature], bins=50, label=feature, kde=False, alpha=0.5)\n\n\n# Add title and labels\nplt.title('Histograms of first 10 Features', fontsize=20)\nplt.xlabel('Value', fontsize=16)\nplt.ylabel('Frequency', fontsize=16)\n\n# Add legend\nplt.legend()\n\n# Show the plot\nplt.tight_layout()\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-11-03T18:22:06.242701Z","iopub.status.idle":"2024-11-03T18:22:06.243151Z","shell.execute_reply.started":"2024-11-03T18:22:06.242937Z","shell.execute_reply":"2024-11-03T18:22:06.242960Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Symbols with highest wieght overtime","metadata":{}},{"cell_type":"code","source":"# Set a Seaborn style for better aesthetics\nsns.set_style(\"whitegrid\")\n\n# Create the figure\nplt.figure(figsize=(16, 6))\n\n# Group and aggregate the data\naggregated = train.groupby('symbol_id')[['weight']].agg(['sum'])\n\n# Create the bar plot\nbars = plt.bar(aggregated.index, aggregated['weight']['sum'], color=sns.color_palette(\"viridis\", len(aggregated)))\n\n# Add title and labels\nplt.title('Total Weight by Symbol ID', fontsize=20)\nplt.xlabel('Symbol ID', fontsize=16)\nplt.ylabel('Total Weight', fontsize=16)\n\n# Rotate x-ticks for better readability\nplt.xticks(rotation=45)\n\n# Optionally, add data labels on top of each bar\nfor bar in bars:\n    yval = bar.get_height()\n    plt.text(bar.get_x() + bar.get_width()/2, yval, round(yval, 2), ha='center', va='bottom', fontsize=10)\n\n# Show the plot with tight layout\nplt.tight_layout()\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-11-03T18:22:06.244957Z","iopub.status.idle":"2024-11-03T18:22:06.245409Z","shell.execute_reply.started":"2024-11-03T18:22:06.245162Z","shell.execute_reply":"2024-11-03T18:22:06.245182Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Distribution of weights over time","metadata":{}},{"cell_type":"code","source":"# Set a Seaborn style for better aesthetics\nsns.set_style(\"whitegrid\")\n\n# Create the figure\nplt.figure(figsize=(16, 6))\n\n# Group and aggregate the data\naggregated = train.groupby('date_id')[['weight']].agg(['sum'])\n\n# Create the bar plot\nbars = plt.bar(aggregated.index, aggregated['weight']['sum'], color=sns.color_palette(\"viridis\", len(aggregated)))\n\n# Add title and labels\nplt.title('Total Weight by Date ID', fontsize=20)\nplt.xlabel('Date ID', fontsize=16)\nplt.ylabel('Total Weight', fontsize=16)\n\n# Rotate x-ticks for better readability\nplt.xticks(rotation=45)\n\n# Show the plot with tight layout\nplt.tight_layout()\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-11-03T18:22:06.247101Z","iopub.status.idle":"2024-11-03T18:22:06.247679Z","shell.execute_reply.started":"2024-11-03T18:22:06.247385Z","shell.execute_reply":"2024-11-03T18:22:06.247414Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"STD of weight is\", train['weight'].std())","metadata":{"execution":{"iopub.status.busy":"2024-11-03T18:32:31.927531Z","iopub.execute_input":"2024-11-03T18:32:31.927969Z","iopub.status.idle":"2024-11-03T18:32:31.972957Z","shell.execute_reply.started":"2024-11-03T18:32:31.927930Z","shell.execute_reply":"2024-11-03T18:32:31.971586Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Outlier Detection","metadata":{}},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns\nimport math\n\n# List of columns to plot boxplots for\ncolumns_to_plot = ['weight','feature_11', 'feature_12', 'feature_13', 'feature_14', 'feature_15', 'feature_16', 'feature_17', 'feature_18', 'feature_19', 'responder_6']\n\n# Set number of columns per row\nn_cols = 3\nn_rows = math.ceil(len(columns_to_plot) / n_cols)\n\n# Create subplots\nfig, axes = plt.subplots(n_rows, n_cols, figsize=(18, 6 * n_rows))\naxes = axes.flatten()  # Flatten axes array to easily index it\n\n# Loop through columns and plot each boxplot in a subplot\nfor i, column in enumerate(columns_to_plot):\n    sns.boxplot(data=train, x=column, ax=axes[i], color='skyblue', flierprops=dict(marker='o', color='red', markersize=7))\n    axes[i].set_title(f\"Boxplot for '{column}'\", fontsize=14)\n    axes[i].set_xlabel(\"Feature Values\", fontsize=12)\n\n# Remove any empty subplots if columns list is not a multiple of n_cols\nfor j in range(i + 1, len(axes)):\n    fig.delaxes(axes[j])\n\n# Adjust layout and show plot\nplt.tight_layout()\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-11-03T18:44:13.337136Z","iopub.execute_input":"2024-11-03T18:44:13.337978Z","iopub.status.idle":"2024-11-03T18:44:25.970415Z","shell.execute_reply.started":"2024-11-03T18:44:13.337925Z","shell.execute_reply":"2024-11-03T18:44:25.969167Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Responder analysis","metadata":{}},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns\n\n# Set a Seaborn style for better aesthetics\nsns.set_style(\"whitegrid\")\n\n# Create the plot for the first 5000 records\nplt.figure(figsize=(16, 6))\n\n# Add title and labels\nplt.title('Responder 6 Over Time (First 5000 Records)', fontsize=20)\nplt.xlabel('Index', fontsize=16)\nplt.ylabel('Responder 6 Value', fontsize=16)\n\nplt.grid(True)\n\n# Optionally, add markers to the line for emphasis\nplt.scatter(range(5000), train['responder_6'][:5000], color='darkblue', s=10, alpha=0.5)\n\n# Show the plot with tight layout\nplt.tight_layout()\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-11-03T18:22:06.251818Z","iopub.status.idle":"2024-11-03T18:22:06.252276Z","shell.execute_reply.started":"2024-11-03T18:22:06.252018Z","shell.execute_reply":"2024-11-03T18:22:06.252039Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['responder_6'].describe()","metadata":{"execution":{"iopub.status.busy":"2024-11-03T18:22:06.253928Z","iopub.status.idle":"2024-11-03T18:22:06.254411Z","shell.execute_reply.started":"2024-11-03T18:22:06.254152Z","shell.execute_reply":"2024-11-03T18:22:06.254174Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Responder Decompostion","metadata":{}},{"cell_type":"code","source":"train[train['responder_6'] > 3][:3]","metadata":{"execution":{"iopub.status.busy":"2024-11-03T18:22:06.256093Z","iopub.status.idle":"2024-11-03T18:22:06.256580Z","shell.execute_reply.started":"2024-11-03T18:22:06.256348Z","shell.execute_reply":"2024-11-03T18:22:06.256371Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"decomposition = seasonal_decompose(train['responder_6'][:1000], model='additive', period=9)\n\n# Create a figure with a specific size\nfig, (ax1, ax2, ax3, ax4) = plt.subplots(4, 1, figsize=(18, 12))  # Change figsize as needed\n\n# Plot the decomposition components\ndecomposition.observed.plot(ax=ax1, legend=False, title='Observed')\ndecomposition.trend.plot(ax=ax2, legend=False, title='Trend')\ndecomposition.seasonal.plot(ax=ax3, legend=False, title='Seasonal')\ndecomposition.resid.plot(ax=ax4, legend=False, title='Residual')\n\n# Show the plot\nplt.tight_layout()\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-11-03T18:22:06.258901Z","iopub.status.idle":"2024-11-03T18:22:06.259349Z","shell.execute_reply.started":"2024-11-03T18:22:06.259096Z","shell.execute_reply":"2024-11-03T18:22:06.259125Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Responder has scaled between [-5, 5] with a \nmean score of -3.7\n\nOverall, the scaling of the responder variable between -5 and 5, along with a mean score of -3.7, suggests a focused distribution of values that leans towards the negative side. This can have significant implications for analysis, modeling, and interpretation within the context of the dataset you are working with. Understanding this distribution can guide further analysis and decision-making based on the responses collected.","metadata":{}},{"cell_type":"code","source":"sns.set_style(\"whitegrid\")\n\n# Create the histogram plot\nplt.figure(figsize=(12, 6))\nsns.histplot(train['responder_6'], bins=50, color='royalblue', kde=True, stat='density', alpha=0.6)\n\n# Add title and labels\nplt.title('Distribution of Responder 6', fontsize=20)\nplt.xlabel('Responder 6 Value', fontsize=16)\nplt.ylabel('Density', fontsize=16)\n\n# Add grid for better readability\nplt.grid(True)\n\n# Show the plot with tight layout\nplt.tight_layout()\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-10-26T08:10:25.273526Z","iopub.execute_input":"2024-10-26T08:10:25.273991Z","iopub.status.idle":"2024-10-26T08:10:57.331001Z","shell.execute_reply.started":"2024-10-26T08:10:25.273924Z","shell.execute_reply":"2024-10-26T08:10:57.329860Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Set a Seaborn style for better aesthetics\nsns.set_style(\"whitegrid\")\n\n# Calculate the cumulative sum\ntrain['cumulative_sum'] = train['responder_6'].cumsum()\n\n# Create the plot\nplt.figure(figsize=(16, 6))\nplt.plot(train['cumulative_sum'], color='royalblue', linewidth=2.5)\n\n# Add title and labels\nplt.title('Cumulative Sum of Responder 6 Over Time', fontsize=20)\nplt.xlabel('Index', fontsize=16)\nplt.ylabel('Cumulative Sum', fontsize=16)\n\n# Add grid for better readability\nplt.grid(True)\n\n# Optionally, add markers to the line for emphasis\nplt.scatter(range(len(train)), train['cumulative_sum'], color='darkblue', s=10, alpha=0.5)\n\n# Show the plot with tight layout\nplt.tight_layout()\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-10-26T08:10:57.332587Z","iopub.execute_input":"2024-10-26T08:10:57.333117Z","iopub.status.idle":"2024-10-26T08:11:07.543863Z","shell.execute_reply.started":"2024-10-26T08:10:57.333061Z","shell.execute_reply":"2024-10-26T08:11:07.542470Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Corrolation Analysis","metadata":{}},{"cell_type":"code","source":"null_cols = []\n\n\n# remove all null columns\nnull_cols = train.columns[pd.isnull(train).all()].tolist()\ntrain = train.drop(null_cols, axis=1)\ntrain.shape\n\nnanValues = train.isnull().sum()\ntrain.dropna(inplace=True)\ntrain.shape","metadata":{"execution":{"iopub.status.busy":"2024-10-26T08:11:07.546529Z","iopub.execute_input":"2024-10-26T08:11:07.546995Z","iopub.status.idle":"2024-10-26T08:11:11.690808Z","shell.execute_reply.started":"2024-10-26T08:11:07.546922Z","shell.execute_reply":"2024-10-26T08:11:11.689671Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Corrolation of features with Responder_6","metadata":{}},{"cell_type":"code","source":"# Drop all responder columns except responder_6\nfeature_columns = train.columns.difference(['responder_0', 'responder_1', 'responder_2', 'responder_3', \n                                         'responder_4', 'responder_5', 'responder_7', 'responder_8', \n                                         'responder_6'])\n\n# Compute the correlation of features with responder_6\ncorr_with_responder6 = train[feature_columns].corrwith(train['responder_6'])\n\n# Visualize the correlation with a bar plot\nplt.figure(figsize=(16, 12))\nsns.barplot(y=corr_with_responder6.index, x=corr_with_responder6.values, palette='coolwarm')\nplt.title('Correlation of Features with responder_6')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-26T08:11:11.692090Z","iopub.execute_input":"2024-10-26T08:11:11.692462Z","iopub.status.idle":"2024-10-26T08:11:25.089954Z","shell.execute_reply.started":"2024-10-26T08:11:11.692423Z","shell.execute_reply":"2024-10-26T08:11:25.088554Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Drop all responder columns except responder_6\ndata = train.drop(columns=['responder_0', 'responder_1', 'responder_2', 'responder_3', \n                        'responder_4', 'responder_5', 'responder_7', 'responder_8'])\n\n# Set up the figure size for a large heatmap\nplt.figure(figsize=(20, 16))\n\n# Generate the heatmap for the dataset correlations\nsns.heatmap(data.corr(), annot=False, cmap='coolwarm', linewidths=0.5)\n\n# Add title\nplt.title('Heatmap of Correlations for All Columns (Including responder_6)')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-24T22:19:23.750750Z","iopub.execute_input":"2024-10-24T22:19:23.751298Z","iopub.status.idle":"2024-10-24T22:21:21.390605Z","shell.execute_reply.started":"2024-10-24T22:19:23.751242Z","shell.execute_reply":"2024-10-24T22:21:21.389474Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data Modeling","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport polars as pl\nfrom xgboost import XGBRegressor\nfrom sklearn.model_selection import train_test_split, GridSearchCV\n\nclass WeightedR2ModelXGB:\n    def __init__(self, data_path, features, target_col, weight_col, split_index):\n        self.data_path = data_path\n        self.features = features\n        self.target_col = target_col\n        self.weight_col = weight_col\n        self.split_index = split_index\n        self.model = XGBRegressor(objective=\"reg:squarederror\", n_jobs=-1)\n\n    def load_data(self):\n        train = pl.read_parquet(self.data_path)\n        train = train.to_pandas()\n        X = train[self.features].fillna(3).values\n        y = train[self.target_col].values\n        weights = train[self.weight_col].values\n        return X, y, weights\n\n    def train_test_split(self, X, y, weights):\n        train_X, train_y = X[:-self.split_index], y[:-self.split_index]\n        test_X, test_y = X[-self.split_index:], y[-self.split_index:]\n        train_weight, test_weight = weights[:-self.split_index], weights[-self.split_index:]\n        return train_X, train_y, test_X, test_y, train_weight, test_weight\n\n    def custom_metric(self, y_true, y_pred, weight):\n        weighted_r2 = 1 - (np.sum(weight * (y_true - y_pred) ** 2) / np.sum(weight * y_true ** 2))\n        return weighted_r2\n\n    def fit(self, train_X, train_y, train_weight):\n        self.model.fit(train_X, train_y, sample_weight=train_weight)\n\n    def predict(self, X):\n        return self.model.predict(X)\n\n    def evaluate(self, train_X, train_y, test_X, test_y, train_weight, test_weight):\n        train_pred = self.predict(train_X)\n        test_pred = self.predict(test_X)\n        train_r2 = self.custom_metric(train_y, train_pred, train_weight)\n        test_r2 = self.custom_metric(test_y, test_pred, test_weight)\n        return train_r2, test_r2\n\n    def run(self):\n        X, y, weights = self.load_data()\n        train_X, train_y, test_X, test_y, train_weight, test_weight = self.train_test_split(X, y, weights)\n        self.fit(train_X, train_y, train_weight)\n        train_r2, test_r2 = self.evaluate(train_X, train_y, test_X, test_y, train_weight, test_weight)\n        return train_r2, test_r2\n\n# Initialize and run the XGBoost model\ndata_path = \"/kaggle/input/jane-street-real-time-market-data-forecasting/train.parquet/partition_id=9/part-0.parquet\"\nfeatures = ['feature_04', 'feature_06', 'feature_07', 'feature_08', 'feature_15', 'feature_16', 'feature_17', 'feature_19', 'feature_25', 'feature_36', 'feature_45', 'feature_56', 'feature_60', 'feature_66']\ntarget_col = 'responder_6'\nweight_col = 'weight'\nsplit_index = 1300000\n\nmodel = WeightedR2ModelXGB(data_path, features, target_col, weight_col, split_index)\ntrain_r2, test_r2 = model.run()\nprint(f\"Train weighted R2: {train_r2}\")\nprint(f\"Test weighted R2: {test_r2}\")","metadata":{"execution":{"iopub.status.busy":"2024-10-26T13:42:39.464712Z","iopub.execute_input":"2024-10-26T13:42:39.465377Z","iopub.status.idle":"2024-10-26T13:43:33.457331Z","shell.execute_reply.started":"2024-10-26T13:42:39.465302Z","shell.execute_reply":"2024-10-26T13:43:33.455758Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def predict(test,lags):\n    cols=['feature_04', 'feature_06', 'feature_07', 'feature_08', 'feature_15', 'feature_16', 'feature_17', 'feature_19', 'feature_25', 'feature_36', 'feature_45', 'feature_56', 'feature_60', 'feature_66']\n    predictions = test.select(\n        'row_id',\n        pl.lit(0.0).alias('responder_6'),\n    )\n    test_preds=model.predict(test[cols].to_pandas().fillna(3).values)\n    predictions = predictions.with_columns(pl.Series('responder_6', test_preds.ravel()))\n    return predictions\n\ninference_server = kaggle_evaluation.jane_street_inference_server.JSInferenceServer(predict)\n\nif os.getenv('KAGGLE_IS_COMPETITION_RERUN'):\n    inference_server.serve()\nelse:\n    inference_server.run_local_gateway(\n        (\n            '/kaggle/input/jane-street-real-time-market-data-forecasting/test.parquet',\n            '/kaggle/input/jane-street-real-time-market-data-forecasting/lags.parquet',\n        )\n    )","metadata":{"execution":{"iopub.status.busy":"2024-10-26T12:29:02.410330Z","iopub.execute_input":"2024-10-26T12:29:02.411532Z","iopub.status.idle":"2024-10-26T12:29:02.458948Z","shell.execute_reply.started":"2024-10-26T12:29:02.411478Z","shell.execute_reply":"2024-10-26T12:29:02.457533Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}