{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":84493,"databundleVersionId":9871156,"sourceType":"competition"}],"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# ","metadata":{}},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-11-27T06:17:59.885052Z","iopub.execute_input":"2024-11-27T06:17:59.885433Z","iopub.status.idle":"2024-11-27T06:18:00.851265Z","shell.execute_reply.started":"2024-11-27T06:17:59.885394Z","shell.execute_reply":"2024-11-27T06:18:00.850416Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\nimport polars as pl\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\nclass JaneStreetDataEDA:\n    def __init__(self, train_path):\n        self.train_path = train_path\n        self.feature_cols = [f'feature_{i:02d}' for i in range(79)]\n        self.responder_cols = [f'responder_{i}' for i in range(9)]\n    \n    def load_data(self):\n        \"\"\"\n        Load all training data partitions\n        \"\"\"\n        all_partitions = []\n        \n        # Iterate through partitions\n        for partition_id in range(4):  # 0 to 9\n            partition_folder = os.path.join(self.train_path, f'partition_id={partition_id}')\n            \n            # Find all parquet files in the partition\n            parquet_files = [\n                os.path.join(partition_folder, f) \n                for f in os.listdir(partition_folder) \n                if f.endswith('.parquet')\n            ]\n            \n            # Load each parquet file in the partition\n            partition_dfs = [pl.read_parquet(f) for f in parquet_files]\n            \n            # Combine files in this partition\n            partition_data = pl.concat(partition_dfs)\n            all_partitions.append(partition_data)\n        \n        # Combine all partitions\n        return pl.concat(all_partitions)\n    \n    def basic_statistics(self, data):\n        \"\"\"\n        Compute basic statistics for features and responders\n        \"\"\"\n        # Convert to pandas for easier analysis\n        df = data.to_pandas()\n        \n        # Basic stats for features\n        feature_stats = df[self.feature_cols].describe()\n        \n        # Basic stats for responders\n        responder_stats = df[self.responder_cols].describe()\n        \n        return {\n            'feature_stats': feature_stats,\n            'responder_stats': responder_stats\n        }\n    \n    def correlation_analysis(self, data):\n        \"\"\"\n        Compute correlation matrices\n        \"\"\"\n        df = data.to_pandas()\n        \n        # Correlation of features\n        feature_corr = df[self.feature_cols].corr()\n        \n        # Correlation of responders\n        responder_corr = df[self.responder_cols].corr()\n        \n        # Correlation between features and responders\n        cross_corr = df[self.feature_cols + self.responder_cols].corr().loc[self.feature_cols, self.responder_cols]\n        \n        return {\n            'feature_correlation': feature_corr,\n            'responder_correlation': responder_corr,\n            'feature_responder_correlation': cross_corr\n        }\n    \n    def distribution_analysis(self, data):\n        \"\"\"\n        Analyze distributions of features and responders\n        \"\"\"\n        df = data.to_pandas()\n        \n        # Prepare plot\n        fig, axes = plt.subplots(5, 3, figsize=(20, 25))\n        axes = axes.flatten()\n        \n        # Plot distributions\n        for i, col in enumerate(self.feature_cols[:15]):\n            sns.histplot(df[col], ax=axes[i], kde=True)\n            axes[i].set_title(f'Distribution of {col}')\n        \n        plt.tight_layout()\n        plt.savefig('feature_distributions.png')\n        plt.close()\n        \n        # Responder distributions\n        fig, axes = plt.subplots(3, 3, figsize=(15, 15))\n        axes = axes.flatten()\n        \n        for i, col in enumerate(self.responder_cols):\n            sns.histplot(df[col], ax=axes[i], kde=True)\n            axes[i].set_title(f'Distribution of {col}')\n        \n        plt.tight_layout()\n        plt.savefig('responder_distributions.png')\n        plt.close()\n    \n    def time_series_characteristics(self, data):\n        \"\"\"\n        Analyze time-based characteristics\n        \"\"\"\n        df = data.to_pandas()\n        \n        # Unique date and time ids\n        date_ids = df['date_id'].unique()\n        time_ids = df['time_id'].unique()\n        symbol_ids = df['symbol_id'].unique()\n        \n        # Time series statistics\n        time_series_stats = {\n            'unique_date_ids': len(date_ids),\n            'unique_time_ids': len(time_ids),\n            'unique_symbol_ids': len(symbol_ids),\n            'date_id_range': (date_ids.min(), date_ids.max()),\n            'time_id_range': (time_ids.min(), time_ids.max()),\n            'symbols_per_date': df.groupby('date_id')['symbol_id'].nunique().describe()\n        }\n        \n        return time_series_stats\n    \n    def weight_analysis(self, data):\n        \"\"\"\n        Analyze weight distribution\n        \"\"\"\n        df = data.to_pandas()\n        \n        weight_stats = {\n            'weight_summary': df['weight'].describe(),\n            'weight_distribution_plot': plt.figure(figsize=(10, 6))\n        }\n        \n        sns.histplot(df['weight'], kde=True)\n        plt.title('Distribution of Weights')\n        plt.savefig('weight_distribution.png')\n        plt.close()\n        \n        return weight_stats\n    \n    def perform_full_eda(self):\n        \"\"\"\n        Comprehensive Exploratory Data Analysis\n        \"\"\"\n        # Load data\n        print(\"Loading data...\")\n        data = self.load_data()\n        \n        # Basic Statistics\n        print(\"\\nComputing Basic Statistics...\")\n        basic_stats = self.basic_statistics(data)\n        \n        # Correlation Analysis\n        print(\"\\nPerforming Correlation Analysis...\")\n        correlations = self.correlation_analysis(data)\n        \n        # Distribution Analysis\n        print(\"\\nAnalyzing Distributions...\")\n        self.distribution_analysis(data)\n        \n        # Time Series Characteristics\n        print(\"\\nAnalyzing Time Series Characteristics...\")\n        time_series_stats = self.time_series_characteristics(data)\n        \n        # Weight Analysis\n        print(\"\\nAnalyzing Weights...\")\n        weight_stats = self.weight_analysis(data)\n        \n        # Generate Report\n        report = {\n            'basic_statistics': basic_stats,\n            'correlations': correlations,\n            'time_series_stats': time_series_stats,\n            'weight_stats': weight_stats\n        }\n        \n        return report\n\n# Main execution\ndef main():\n    # Update this path to your train.parquet folder\n    train_path = '/kaggle/input/jane-street-real-time-market-data-forecasting/train.parquet'\n    \n    # Initialize EDA\n    eda = JaneStreetDataEDA(train_path)\n    \n    # Perform full EDA\n    report = eda.perform_full_eda()\n    \n    # Print key insights\n    print(\"\\n--- EDA Report Highlights ---\")\n    print(\"\\nBasic Feature Statistics:\")\n    print(report['basic_statistics']['feature_stats'])\n    \n    print(\"\\nTime Series Characteristics:\")\n    for key, value in report['time_series_stats'].items():\n        print(f\"{key}: {value}\")\n    \n    # Save detailed report\n    import json\n    with open('jane_street_eda_report.json', 'w') as f:\n        json.dump({k: str(v) for k, v in report.items()}, f, indent=2)\n    \n    print(\"\\nFull EDA report saved to jane_street_eda_report.json\")\n    print(\"Distribution plots saved as PNG files\")\n\nif __name__ == '__main__':\n    main()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T06:18:00.852853Z","iopub.execute_input":"2024-11-27T06:18:00.853205Z","iopub.status.idle":"2024-11-27T06:46:47.675183Z","shell.execute_reply.started":"2024-11-27T06:18:00.853178Z","shell.execute_reply":"2024-11-27T06:46:47.674171Z"}},"outputs":[],"execution_count":null}]}