{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84493,"databundleVersionId":9871156,"sourceType":"competition"}],"dockerImageVersionId":30786,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import polars as pl\n\n# Initialize an empty list to hold dataframes\ndataframes = []\n\n# Read all 10 parquet files\nfor i in range(1):\n    file_path = f'/kaggle/input/jane-street-real-time-market-data-forecasting/train.parquet/partition_id={i}/'\n    df = pl.read_parquet(file_path)\n    dataframes.append(df)\n\n# Concatenate all dataframes\ndata = pl.concat(dataframes)\n\n# Convert to pandas for EDA\ndata_pd = data.to_pandas()\n\n# Display the first few rows of the dataframe\nprint(data_pd.head())\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-10-27T12:26:32.676613Z","iopub.execute_input":"2024-10-27T12:26:32.67704Z","iopub.status.idle":"2024-10-27T12:26:36.225439Z","shell.execute_reply.started":"2024-10-27T12:26:32.676999Z","shell.execute_reply":"2024-10-27T12:26:36.223888Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"nan_counts = data_pd.isnull().sum()\nprint(nan_counts[nan_counts > 0])  # Only show features with NaN values\n","metadata":{"execution":{"iopub.status.busy":"2024-10-27T12:26:40.167495Z","iopub.execute_input":"2024-10-27T12:26:40.167942Z","iopub.status.idle":"2024-10-27T12:26:40.428143Z","shell.execute_reply.started":"2024-10-27T12:26:40.167898Z","shell.execute_reply":"2024-10-27T12:26:40.426964Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns\nfrom scipy.stats import skew, kurtosis\n\nfor responder in [f'responder_{i}' for i in range(9)]:\n    plt.figure(figsize=(12, 6))\n    sns.histplot(data_pd[responder].dropna(), bins=30, kde=True)\n    plt.title(f'Distribution of {responder}')\n    plt.xlabel(responder)\n    plt.ylabel('Frequency')\n    plt.show()\n\n    # Calculate skewness and kurtosis\n    print(f'{responder} - Skewness: {skew(data_pd[responder].dropna())}, Kurtosis: {kurtosis(data_pd[responder].dropna())}')\n","metadata":{"execution":{"iopub.status.busy":"2024-10-27T12:26:43.598002Z","iopub.execute_input":"2024-10-27T12:26:43.598419Z","iopub.status.idle":"2024-10-27T12:28:09.253772Z","shell.execute_reply.started":"2024-10-27T12:26:43.598376Z","shell.execute_reply":"2024-10-27T12:28:09.252633Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns\n\n# Calculate the correlation matrix\ncorrelation_matrix = data_pd.corr()\n\n# Set up the matplotlib figure\nplt.figure(figsize=(14, 12))\n\n# Create a heatmap without annotations\nsns.heatmap(correlation_matrix, cmap='coolwarm', square=True, cbar_kws={\"shrink\": .8}, annot=False)\n\n# Set title and labels\nplt.title('Correlation Heatmap', fontsize=16)\nplt.xticks(rotation=45, ha='right')\nplt.yticks(rotation=0)\nplt.tight_layout()\n\n# Show the plot\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-10-27T12:28:09.255716Z","iopub.execute_input":"2024-10-27T12:28:09.256311Z","iopub.status.idle":"2024-10-27T12:28:48.781039Z","shell.execute_reply.started":"2024-10-27T12:28:09.25627Z","shell.execute_reply":"2024-10-27T12:28:48.779939Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns\n\n# Select only the responder columns\nresponders = data_pd[[f'responder_{i}' for i in range(9)]]\n\n# Calculate the correlation matrix for responders\ncorrelation_matrix_responders = responders.corr()\n\n# Set up the matplotlib figure\nplt.figure(figsize=(10, 8))\n\n# Create a heatmap without annotations\nsns.heatmap(correlation_matrix_responders, cmap='coolwarm', square=True, cbar_kws={\"shrink\": .8}, annot=False)\n\n# Set title and labels\nplt.title('Correlation Heatmap of Responders', fontsize=16)\nplt.xticks(rotation=45, ha='right')\nplt.yticks(rotation=0)\nplt.tight_layout()\n\n# Show the plot\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-10-27T12:28:48.782334Z","iopub.execute_input":"2024-10-27T12:28:48.782672Z","iopub.status.idle":"2024-10-27T12:28:49.804271Z","shell.execute_reply.started":"2024-10-27T12:28:48.782637Z","shell.execute_reply":"2024-10-27T12:28:49.802955Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nfrom statsmodels.tsa.stattools import adfuller\nimport matplotlib.pyplot as plt\n\n# Define the ADF test function for pandas Series\ndef adf_test(series, title=''):\n    \"\"\"\n    Perform the Augmented Dickey-Fuller test to check for stationarity.\n    \n    :param series: Pandas Series - time series to test\n    :param title: str - title for the series to identify output\n    \"\"\"\n    result = adfuller(series.dropna(), autolag='AIC')  # Ensuring no NaN values\n    print(f'Results of Dickey-Fuller Test: {title}')\n    print('-----------------------------------')\n    print('Test Statistic:', result[0])\n    print('p-value:', result[1])\n    print('Number of Lags Used:', result[2])\n    print('Number of Observations Used:', result[3])\n    print('Critical Values:')\n    for key, value in result[4].items():\n        print(f'    {key}: {value}')\n    print('-----------------------------------\\n')\n\n# Apply the ADF test to each responder directly from pandas DataFrame\nfor i in range(6):  # Assuming you have responders from responder_0 to responder_5\n    series_pd = data_pd[f'responder_{i}'].dropna()  # Directly using pandas Series\n    adf_test(series_pd, title=f'responder_{i}')\n\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}