{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":96164,"databundleVersionId":11418275,"sourceType":"competition"}],"dockerImageVersionId":31040,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-05-25T20:37:56.993961Z","iopub.execute_input":"2025-05-25T20:37:56.999005Z","iopub.status.idle":"2025-05-25T20:37:59.522633Z","shell.execute_reply.started":"2025-05-25T20:37:56.998731Z","shell.execute_reply":"2025-05-25T20:37:59.521319Z"},"jupyter":{"source_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\n\ndef diagnose_data_structure(train_path='train.parquet', test_path='test.parquet'):\n    \"\"\"\n    Quick diagnostic to understand the data structure\n    \"\"\"\n    print(\"=== DATA STRUCTURE DIAGNOSTIC ===\\n\")\n    \n    # Load a small sample first\n    print(\"Loading sample data...\")\n    train_df = pd.read_parquet(train_path)\n    \n    print(f\"Training data shape: {train_df.shape}\")\n    print(f\"\\nColumn names (first 20):\")\n    for i, col in enumerate(train_df.columns[:20]):\n        print(f\"  {i}: {col}\")\n    \n    print(f\"\\n... and {len(train_df.columns) - 20} more columns\")\n    \n    # Check for timestamp column\n    timestamp_cols = [col for col in train_df.columns if 'time' in col.lower() or 'date' in col.lower()]\n    print(f\"\\nPotential timestamp columns: {timestamp_cols}\")\n    \n    # Check data types\n    print(\"\\nData types summary:\")\n    dtype_counts = train_df.dtypes.value_counts()\n    for dtype, count in dtype_counts.items():\n        print(f\"  {dtype}: {count} columns\")\n    \n    # Check for label column\n    if 'label' in train_df.columns:\n        print(f\"\\nLabel column found!\")\n        print(f\"  Label range: [{train_df['label'].min():.4f}, {train_df['label'].max():.4f}]\")\n        print(f\"  Label mean: {train_df['label'].mean():.4f}\")\n        print(f\"  Label std: {train_df['label'].std():.4f}\")\n    else:\n        print(\"\\nWarning: 'label' column not found!\")\n    \n    # Check market features\n    market_features = ['bid_qty', 'ask_qty', 'buy_qty', 'sell_qty', 'volume']\n    found_market_features = [f for f in market_features if f in train_df.columns]\n    print(f\"\\nMarket features found: {found_market_features}\")\n    \n    # Check anonymous features\n    anon_features = [col for col in train_df.columns if col.startswith('X_')]\n    print(f\"\\nAnonymous features (X_*): {len(anon_features)} found\")\n    \n    # Sample first few rows\n    print(\"\\nFirst 5 rows of key columns:\")\n    display_cols = found_market_features[:3] + ['label'] if 'label' in train_df.columns else found_market_features[:3]\n    if display_cols:\n        print(train_df[display_cols].head())\n    \n    # Check for missing values\n    print(\"\\nMissing values summary:\")\n    missing_counts = train_df.isnull().sum()\n    missing_cols = missing_counts[missing_counts > 0]\n    if len(missing_cols) > 0:\n        print(f\"  Columns with missing values: {len(missing_cols)}\")\n        print(f\"  Top 5 columns with most missing values:\")\n        for col, count in missing_cols.nlargest(5).items():\n            print(f\"    {col}: {count} ({count/len(train_df)*100:.1f}%)\")\n    else:\n        print(\"  No missing values found!\")\n    \n    # Quick temporal analysis without timestamp\n    print(\"\\n=== TEMPORAL ANALYSIS (based on row order) ===\")\n    \n    # Plot label evolution\n    fig, axes = plt.subplots(3, 1, figsize=(12, 10))\n    \n    # 1. Label values over time\n    ax = axes[0]\n    sample_size = min(len(train_df), 50000)  # Plot subset for efficiency\n    indices = np.linspace(0, len(train_df)-1, sample_size).astype(int)\n    ax.plot(indices, train_df['label'].iloc[indices], alpha=0.6, linewidth=0.5)\n    ax.set_title('Label Evolution (assuming chronological order)')\n    ax.set_xlabel('Row Index')\n    ax.set_ylabel('Label Value')\n    ax.grid(True, alpha=0.3)\n    \n    # 2. Rolling statistics\n    ax = axes[1]\n    window = 10000  # 10k minute window\n    rolling_mean = train_df['label'].rolling(window=window, min_periods=1).mean()\n    rolling_std = train_df['label'].rolling(window=window, min_periods=1).std()\n    \n    ax.plot(indices, rolling_mean.iloc[indices], label='Rolling Mean', alpha=0.8)\n    ax.fill_between(indices, \n                     (rolling_mean - 2*rolling_std).iloc[indices],\n                     (rolling_mean + 2*rolling_std).iloc[indices],\n                     alpha=0.2, label='±2 Std Dev')\n    ax.set_title(f'Label Rolling Statistics ({window} row window)')\n    ax.set_xlabel('Row Index')\n    ax.set_ylabel('Label Value')\n    ax.legend()\n    ax.grid(True, alpha=0.3)\n    \n    # 3. Volume evolution (if available)\n    ax = axes[2]\n    if 'volume' in train_df.columns:\n        volume_rolling = train_df['volume'].rolling(window=window, min_periods=1).mean()\n        ax.plot(indices, volume_rolling.iloc[indices], color='green', alpha=0.8)\n        ax.set_title(f'Volume Evolution ({window} row rolling average)')\n        ax.set_xlabel('Row Index')\n        ax.set_ylabel('Volume')\n        ax.grid(True, alpha=0.3)\n    else:\n        ax.text(0.5, 0.5, 'Volume data not available', \n                horizontalalignment='center', verticalalignment='center',\n                transform=ax.transAxes)\n    \n    plt.tight_layout()\n    plt.savefig('data_diagnostic_plots.png', dpi=150, bbox_inches='tight')\n    plt.show()\n    \n    # Correlation analysis\n    print(\"\\n=== QUICK CORRELATION ANALYSIS ===\")\n    \n    # Sample for efficiency\n    sample_df = train_df.sample(n=min(10000, len(train_df)), random_state=42)\n    \n    # Correlations with label\n    if 'label' in sample_df.columns:\n        # Market features correlation\n        if found_market_features:\n            print(\"\\nMarket features correlation with label:\")\n            for feature in found_market_features:\n                corr = sample_df[feature].corr(sample_df['label'])\n                print(f\"  {feature}: {corr:.4f}\")\n        \n        # Top anonymous features by correlation\n        if anon_features:\n            print(\"\\nTop 10 anonymous features by absolute correlation with label:\")\n            anon_corrs = {}\n            for feature in anon_features[:100]:  # Check first 100 for efficiency\n                corr = sample_df[feature].corr(sample_df['label'])\n                anon_corrs[feature] = abs(corr)\n            \n            top_features = sorted(anon_corrs.items(), key=lambda x: x[1], reverse=True)[:10]\n            for feature, corr in top_features:\n                actual_corr = sample_df[feature].corr(sample_df['label'])\n                print(f\"  {feature}: {actual_corr:.4f} (abs: {corr:.4f})\")\n    \n    print(\"\\n=== DIAGNOSTIC COMPLETE ===\")\n    return train_df\n\n# Run diagnostic\nif __name__ == \"__main__\":\n    train_path = '/kaggle/input/drw-crypto-market-prediction/train.parquet'\n    df = diagnose_data_structure(train_path)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T20:37:59.534591Z","iopub.execute_input":"2025-05-25T20:37:59.535224Z","iopub.status.idle":"2025-05-25T20:38:53.250524Z","shell.execute_reply.started":"2025-05-25T20:37:59.535184Z","shell.execute_reply":"2025-05-25T20:38:53.249195Z"},"jupyter":{"source_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Load and check the test set size\ntest_df = pd.read_parquet('/kaggle/input/drw-crypto-market-prediction/test.parquet')\nprint(f\"Test set shape: {test_df.shape}\")\nprint(f\"Number of test records: {len(test_df)}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T20:38:53.252247Z","iopub.execute_input":"2025-05-25T20:38:53.252665Z","iopub.status.idle":"2025-05-25T20:39:32.027270Z","shell.execute_reply.started":"2025-05-25T20:38:53.252631Z","shell.execute_reply":"2025-05-25T20:39:32.025360Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom prophet import Prophet\nimport gc\nimport warnings\nwarnings.filterwarnings('ignore')\n\nplt.style.use('seaborn-v0_8-whitegrid')\n\nclass MemoryEfficientProphetAnalyzer:\n    \"\"\"\n    Memory-efficient Prophet analyzer for large crypto datasets.\n    \"\"\"\n    \n    def __init__(self, data_path):\n        self.data_path = data_path\n        self.regime_df = None\n        self.changepoints = None\n        \n    def load_minimal_data(self, columns=['timestamp', 'label', 'volume']):\n        \"\"\"\n        Load only essential columns to save memory.\n        \"\"\"\n        print(\"Loading minimal data...\")\n        \n        # Read only specific columns\n        df = pd.read_parquet(self.data_path, columns=columns)\n        \n        # Ensure timestamp exists\n        if 'timestamp' not in df.columns:\n            if df.index.name == 'timestamp':\n                df = df.reset_index()\n            else:\n                start_date = pd.Timestamp('2023-03-01')\n                df['timestamp'] = pd.date_range(start=start_date, periods=len(df), freq='T')\n        \n        print(f\"Loaded {len(df):,} rows with columns: {list(df.columns)}\")\n        \n        # Clean data\n        for col in df.select_dtypes(include=[np.number]).columns:\n            df[col] = df[col].replace([np.inf, -np.inf], np.nan).fillna(df[col].median())\n        \n        return df\n    \n    def aggressive_resample_for_prophet(self, df, resample_freq='4H'):\n        \"\"\"\n        Aggressively resample data to reduce memory usage.\n        \"\"\"\n        print(f\"Resampling data to {resample_freq} intervals...\")\n        \n        # Resample to reduce data size\n        prophet_df = pd.DataFrame({\n            'ds': df['timestamp'],\n            'y': df['label']\n        })\n        \n        # Set index and resample\n        prophet_df = prophet_df.set_index('ds').resample(resample_freq).agg({\n            'y': ['mean', 'std', 'min', 'max']\n        })\n        \n        # Flatten column names\n        prophet_df.columns = ['_'.join(col).strip() for col in prophet_df.columns]\n        \n        # Use mean for Prophet, keep others for analysis\n        prophet_input = pd.DataFrame({\n            'ds': prophet_df.index,\n            'y': prophet_df['y_mean']\n        }).reset_index(drop=True)\n        \n        # Remove NaN values\n        prophet_input = prophet_input.dropna()\n        \n        print(f\"Resampled to {len(prophet_input):,} data points\")\n        \n        # Store additional stats for later use\n        self.resampled_stats = prophet_df\n        \n        # Clean up\n        del prophet_df\n        gc.collect()\n        \n        return prophet_input\n    \n    def detect_regimes_memory_efficient(self, df, resample='4H', n_changepoints=10):\n        \"\"\"\n        Detect regimes with minimal memory usage.\n        \"\"\"\n        print(\"\\nDetecting regimes with memory optimization...\")\n        \n        # Resample data\n        prophet_df = self.aggressive_resample_for_prophet(df, resample)\n        \n        if len(prophet_df) < 20:\n            print(\"Too few data points after resampling\")\n            return None, None, []\n        \n        # Configure Prophet for minimal memory usage\n        model = Prophet(\n            changepoint_prior_scale=0.05,\n            n_changepoints=min(n_changepoints, len(prophet_df) // 20),\n            yearly_seasonality=False,\n            weekly_seasonality=False,  # Disable to save memory\n            daily_seasonality=False,   # Disable to save memory\n            seasonality_mode='additive',  # Simpler than multiplicative\n            mcmc_samples=0,  # Disable MCMC for memory\n            interval_width=0.8  # Reduce from 0.95 to save memory\n        )\n        \n        # Fit model\n        print(\"Fitting Prophet model...\")\n        model.fit(prophet_df)\n        \n        # Generate minimal forecast\n        future = model.make_future_dataframe(periods=0)\n        forecast = model.predict(future)\n        \n        # Extract changepoints and clean up\n        changepoints = model.changepoints.copy()\n        \n        # Map changepoints to original data indices\n        changepoint_indices = []\n        for cp in changepoints:\n            # Find closest timestamp in original data\n            closest_idx = np.argmin(np.abs(df['timestamp'] - pd.Timestamp(cp)))\n            changepoint_indices.append(closest_idx)\n        \n        changepoint_indices = sorted(list(set(changepoint_indices)))\n        \n        # Clean up Prophet objects\n        del prophet_df\n        gc.collect()\n        \n        return model, forecast, changepoint_indices\n    \n    def analyze_regimes_minimal(self, df, changepoint_indices):\n        \"\"\"\n        Analyze regimes with minimal memory footprint.\n        \"\"\"\n        if not changepoint_indices:\n            return pd.DataFrame()\n        \n        print(\"Analyzing regime characteristics...\")\n        \n        regime_boundaries = [0] + changepoint_indices + [len(df)]\n        regimes = []\n        \n        for i in range(len(regime_boundaries) - 1):\n            start = regime_boundaries[i]\n            end = regime_boundaries[i + 1]\n            \n            # Process in chunks if regime is large\n            if end - start > 100000:\n                # Sample the regime instead of using all data\n                sample_size = min(50000, end - start)\n                indices = np.random.choice(range(start, end), sample_size, replace=False)\n                regime_data = df.iloc[sorted(indices)]\n            else:\n                regime_data = df.iloc[start:end]\n            \n            # Calculate basic statistics\n            regime_info = {\n                'regime_id': i,\n                'start_date': df['timestamp'].iloc[start],\n                'end_date': df['timestamp'].iloc[end-1] if end < len(df) else df['timestamp'].iloc[-1],\n                'duration_hours': (end - start) / 60,\n                'n_samples': end - start,\n                'label_mean': regime_data['label'].mean(),\n                'label_std': regime_data['label'].std(),\n                'label_min': regime_data['label'].min(),\n                'label_max': regime_data['label'].max()\n            }\n            \n            # Add volume if available\n            if 'volume' in regime_data.columns:\n                regime_info['volume_mean'] = regime_data['volume'].mean()\n            \n            regimes.append(regime_info)\n            \n            # Clean up\n            del regime_data\n            gc.collect()\n        \n        return pd.DataFrame(regimes)\n    \n    def create_memory_efficient_plot(self, df, model, forecast, changepoint_indices, regime_df):\n        \"\"\"\n        Create visualizations with memory efficiency in mind.\n        \"\"\"\n        print(\"Creating memory-efficient visualizations...\")\n        \n        fig, axes = plt.subplots(2, 2, figsize=(15, 10))\n        \n        # 1. Downsampled regime plot\n        ax = axes[0, 0]\n        \n        # Downsample for plotting\n        plot_sample = min(10000, len(df))\n        if len(df) > plot_sample:\n            plot_indices = np.linspace(0, len(df)-1, plot_sample, dtype=int)\n            plot_df = df.iloc[plot_indices]\n        else:\n            plot_df = df\n        \n        ax.plot(plot_df['timestamp'], plot_df['label'], 'b-', alpha=0.6, linewidth=1)\n        \n        # Add changepoints\n        for idx in changepoint_indices:\n            if idx < len(df):\n                ax.axvline(x=df['timestamp'].iloc[idx], color='red', \n                          linestyle='--', alpha=0.7, linewidth=2)\n        \n        ax.set_title('Market Regimes (Downsampled)', fontsize=12, fontweight='bold')\n        ax.set_xlabel('Date')\n        ax.set_ylabel('Label')\n        ax.grid(True, alpha=0.3)\n        \n        # 2. Regime summary\n        ax = axes[0, 1]\n        \n        if not regime_df.empty:\n            # Simple bar chart of regime durations\n            ax.bar(regime_df['regime_id'], regime_df['duration_hours'], \n                   color='skyblue', edgecolor='navy', linewidth=1)\n            ax.set_title('Regime Durations', fontsize=12, fontweight='bold')\n            ax.set_xlabel('Regime ID')\n            ax.set_ylabel('Duration (hours)')\n            ax.grid(True, alpha=0.3, axis='y')\n        \n        # 3. Volatility by regime\n        ax = axes[1, 0]\n        \n        if not regime_df.empty and 'label_std' in regime_df:\n            ax.bar(regime_df['regime_id'], regime_df['label_std'], \n                   color='coral', edgecolor='darkred', linewidth=1)\n            ax.set_title('Regime Volatility', fontsize=12, fontweight='bold')\n            ax.set_xlabel('Regime ID')\n            ax.set_ylabel('Label Std Dev')\n            ax.grid(True, alpha=0.3, axis='y')\n        \n        # 4. Summary text\n        ax = axes[1, 1]\n        ax.axis('off')\n        \n        summary_text = \"REGIME SUMMARY\\n\" + \"=\"*25 + \"\\n\\n\"\n        \n        if not regime_df.empty:\n            summary_text += f\"Total Regimes: {len(regime_df)}\\n\"\n            summary_text += f\"Avg Duration: {regime_df['duration_hours'].mean():.1f} hours\\n\"\n            summary_text += f\"Max Duration: {regime_df['duration_hours'].max():.1f} hours\\n\\n\"\n            \n            if 'label_std' in regime_df:\n                summary_text += f\"Volatility Range:\\n\"\n                summary_text += f\"  Min: {regime_df['label_std'].min():.4f}\\n\"\n                summary_text += f\"  Max: {regime_df['label_std'].max():.4f}\\n\\n\"\n            \n            # Current regime\n            current = regime_df.iloc[-1]\n            summary_text += f\"Current Regime (R{current['regime_id']}):\\n\"\n            summary_text += f\"  Duration: {current['duration_hours']:.1f}h\\n\"\n            summary_text += f\"  Mean: {current['label_mean']:.4f}\\n\"\n            summary_text += f\"  Volatility: {current['label_std']:.4f}\\n\"\n        \n        ax.text(0.1, 0.9, summary_text, transform=ax.transAxes, \n               fontsize=11, verticalalignment='top', fontfamily='monospace',\n               bbox=dict(boxstyle='round', facecolor='wheat', alpha=0.8))\n        \n        plt.tight_layout()\n        plt.savefig('prophet_regime_memory_efficient.png', dpi=150, bbox_inches='tight')\n        plt.show()\n        \n        # Clean up\n        plt.close()\n        gc.collect()\n    \n    def export_regime_labels(self, df, changepoint_indices, output_file='regime_labels.parquet'):\n        \"\"\"\n        Export regime labels efficiently.\n        \"\"\"\n        print(\"Exporting regime labels...\")\n        \n        regime_boundaries = [0] + changepoint_indices + [len(df)]\n        \n        # Create regime labels array\n        regime_labels = np.zeros(len(df), dtype=np.int8)  # Use int8 to save memory\n        \n        for i in range(len(regime_boundaries) - 1):\n            start = regime_boundaries[i]\n            end = regime_boundaries[i + 1]\n            regime_labels[start:end] = i\n        \n        # Create minimal output dataframe\n        output_df = pd.DataFrame({\n            'timestamp': df['timestamp'],\n            'regime_id': regime_labels\n        })\n        \n        # Add regime statistics\n        if self.regime_df is not None:\n            regime_stats = self.regime_df[['regime_id', 'label_mean', 'label_std']].copy()\n            output_df = output_df.merge(regime_stats, on='regime_id', how='left')\n        \n        # Save as parquet for efficiency\n        output_df.to_parquet(output_file, compression='snappy')\n        print(f\"Saved regime labels to {output_file}\")\n        \n        return output_df\n\n\ndef run_memory_efficient_analysis(data_path='/kaggle/input/drw-crypto-market-prediction/train.parquet'):\n    \"\"\"\n    Run memory-efficient Prophet analysis.\n    \"\"\"\n    print(\"=\"*60)\n    print(\"MEMORY-EFFICIENT PROPHET REGIME ANALYSIS\")\n    print(\"=\"*60)\n    \n    # Initialize analyzer\n    analyzer = MemoryEfficientProphetAnalyzer(data_path)\n    \n    # Load minimal data\n    df = analyzer.load_minimal_data(columns=['timestamp', 'label', 'volume'])\n    \n    # Run regime detection\n    model, forecast, changepoints = analyzer.detect_regimes_memory_efficient(\n        df, \n        resample='4H',  # Aggressive resampling\n        n_changepoints=10  # Fewer changepoints\n    )\n    \n    if not changepoints:\n        print(\"No changepoints detected\")\n        return analyzer, pd.DataFrame()\n    \n    print(f\"\\nDetected {len(changepoints)} changepoints\")\n    \n    # Analyze regimes\n    regime_df = analyzer.analyze_regimes_minimal(df, changepoints)\n    analyzer.regime_df = regime_df\n    analyzer.changepoints = changepoints\n    \n    # Create visualization\n    analyzer.create_memory_efficient_plot(df, model, forecast, changepoints, regime_df)\n    \n    # Export labels\n    regime_labels_df = analyzer.export_regime_labels(df, changepoints)\n    \n    # Print summary\n    print(\"\\n\" + \"=\"*60)\n    print(\"ANALYSIS COMPLETE\")\n    print(\"=\"*60)\n    print(f\"\\nKey Results:\")\n    print(f\"- {len(regime_df)} regimes detected\")\n    print(f\"- Average regime duration: {regime_df['duration_hours'].mean():.1f} hours\")\n    print(f\"- Output saved to: prophet_regime_memory_efficient.png\")\n    print(f\"- Regime labels saved to: regime_labels.parquet\")\n    \n    # Clean up\n    del df\n    del model\n    del forecast\n    gc.collect()\n    \n    return analyzer, regime_df\n\n\n# Alternative: Ultra-light version for extreme memory constraints\ndef ultra_light_prophet_analysis(data_path):\n    \"\"\"\n    Ultra-light version that processes data in chunks.\n    \"\"\"\n    print(\"Running ultra-light Prophet analysis...\")\n    \n    # Read only timestamp and label\n    df = pd.read_parquet(data_path, columns=['label'])\n    \n    # Create timestamp if needed\n    if 'timestamp' not in df.columns:\n        df['timestamp'] = pd.date_range('2023-03-01', periods=len(df), freq='T')\n    \n    # Extreme downsampling - daily averages\n    daily_df = df.set_index('timestamp').resample('D').agg({\n        'label': ['mean', 'std', 'count']\n    })\n    \n    daily_df.columns = ['label_mean', 'label_std', 'count']\n    daily_df = daily_df.reset_index()\n    \n    # Simple Prophet model\n    prophet_df = pd.DataFrame({\n        'ds': daily_df['timestamp'],\n        'y': daily_df['label_mean']\n    }).dropna()\n    \n    model = Prophet(\n        changepoint_prior_scale=0.1,\n        n_changepoints=5,\n        yearly_seasonality=False,\n        weekly_seasonality=False,\n        daily_seasonality=False\n    )\n    \n    model.fit(prophet_df)\n    \n    # Get changepoints\n    changepoints = model.changepoints\n    \n    print(f\"Found {len(changepoints)} major regime changes\")\n    print(\"\\nChangepoint dates:\")\n    for cp in changepoints:\n        print(f\"  - {cp.strftime('%Y-%m-%d')}\")\n    \n    # Simple plot\n    fig, ax = plt.subplots(figsize=(12, 6))\n    ax.plot(daily_df['timestamp'], daily_df['label_mean'], 'b-', alpha=0.7)\n    \n    for cp in changepoints:\n        ax.axvline(x=cp, color='red', linestyle='--', alpha=0.7)\n    \n    ax.set_title('Daily Average Regimes')\n    ax.set_xlabel('Date')\n    ax.set_ylabel('Label (Daily Average)')\n    ax.grid(True, alpha=0.3)\n    \n    plt.tight_layout()\n    plt.savefig('ultra_light_regimes.png', dpi=100)\n    plt.show()\n    \n    return model, changepoints\n\n\nif __name__ == \"__main__\":\n    # Try memory-efficient version first\n    try:\n        analyzer, regime_df = run_memory_efficient_analysis()\n    except MemoryError:\n        print(\"\\nMemory error encountered. Trying ultra-light version...\")\n        model, changepoints = ultra_light_prophet_analysis(\n            '/kaggle/input/drw-crypto-market-prediction/train.parquet'\n        )","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T20:39:32.028794Z","iopub.execute_input":"2025-05-25T20:39:32.029137Z","iopub.status.idle":"2025-05-25T20:39:40.409983Z","shell.execute_reply.started":"2025-05-25T20:39:32.029110Z","shell.execute_reply":"2025-05-25T20:39:40.408604Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom sklearn.preprocessing import StandardScaler, RobustScaler\nfrom sklearn.ensemble import RandomForestRegressor, GradientBoostingRegressor, HistGradientBoostingRegressor\nfrom sklearn.metrics import mean_squared_error, mean_absolute_error, r2_score\nfrom scipy import stats\nfrom scipy.stats import pearsonr, spearmanr\nfrom scipy.optimize import curve_fit\nimport lightgbm as lgb\nfrom datetime import datetime, timedelta\nimport warnings\nwarnings.filterwarnings('ignore')\n\n# Set professional visualization style\nplt.style.use('seaborn-v0_8-whitegrid')\nsns.set_palette(\"deep\")\n\nclass TemporalValidationAnalyzer:\n    \"\"\"\n    Framework for analyzing temporal stability and performance degradation\n    in predictive models across different time periods.\n    \"\"\"\n    \n    def __init__(self, data_path='train.parquet'):\n        self.data_path = data_path\n        self.df = None\n        self.results = {}\n        self.feature_importance_evolution = {}\n        \n    def load_and_prepare_data(self):\n        \"\"\"\n        Load data with robust preprocessing for temporal analysis.\n        \"\"\"\n        print(\"Loading and preparing data for temporal validation analysis...\")\n        \n        self.df = pd.read_parquet(self.data_path)\n        \n        # Handle timestamp creation if needed\n        if 'timestamp' not in self.df.columns and self.df.index.name == 'timestamp':\n            self.df = self.df.reset_index()\n        elif 'timestamp' not in self.df.columns:\n            start_date = pd.Timestamp('2023-03-01')\n            self.df['timestamp'] = pd.date_range(start=start_date, periods=len(self.df), freq='T')\n        \n        # Handle extreme values and infinities\n        numeric_cols = self.df.select_dtypes(include=[np.number]).columns\n        for col in numeric_cols:\n            if col != 'label':\n                self.df[col] = self.df[col].replace([np.inf, -np.inf], np.nan)\n                upper_cap = self.df[col].quantile(0.9999)\n                lower_cap = self.df[col].quantile(0.0001)\n                self.df[col] = self.df[col].clip(lower=lower_cap, upper=upper_cap)\n                self.df[col] = self.df[col].fillna(self.df[col].median())\n        \n        print(f\"Data prepared. Shape: {self.df.shape}\")\n        print(f\"Date range: {self.df['timestamp'].min()} to {self.df['timestamp'].max()}\")\n        \n        return self.df\n    \n    def perform_temporal_cross_validation(self, window_size=50000, n_splits=10):\n        \"\"\"\n        Perform sophisticated temporal cross-validation to analyze performance degradation.\n        \"\"\"\n        print(f\"\\nPerforming temporal cross-validation with {window_size:,} row windows...\")\n        \n        # Define feature columns\n        feature_cols = [col for col in self.df.columns if col not in ['timestamp', 'label']]\n        \n        # Initialize results storage\n        validation_results = []\n        \n        # Generate temporal splits\n        total_rows = len(self.df)\n        step_size = (total_rows - 2 * window_size) // (n_splits - 1)\n        \n        for split_idx in range(n_splits):\n            print(f\"\\nProcessing split {split_idx + 1}/{n_splits}\")\n            \n            # Define training window (recent data)\n            train_end = total_rows - (split_idx * step_size)\n            train_start = train_end - window_size\n            \n            if train_start < 0:\n                continue\n            \n            # Extract training data\n            X_train = self.df[feature_cols].iloc[train_start:train_end].values\n            y_train = self.df['label'].iloc[train_start:train_end].values\n            train_timestamp = self.df['timestamp'].iloc[train_end - 1]\n            \n            # Scale features\n            scaler = RobustScaler()\n            X_train_scaled = scaler.fit_transform(X_train)\n            \n            # Train model\n            model = lgb.LGBMRegressor(\n                n_estimators=100,\n                max_depth=5,\n                learning_rate=0.1,\n                subsample=0.8,\n                colsample_bytree=0.8,\n                random_state=42,\n                n_jobs=-1,\n                verbose=-1\n            )\n            \n            model.fit(X_train_scaled, y_train)\n            \n            # Store feature importances\n            feature_importances = model.feature_importances_\n            \n            # Test on multiple historical windows\n            for lookback_periods in [0, 1, 2, 3, 4, 5]:\n                test_end = train_start - (lookback_periods * window_size)\n                test_start = test_end - 10000  # Test on 10k rows\n                \n                if test_start < 0:\n                    continue\n                \n                # Extract test data\n                X_test = self.df[feature_cols].iloc[test_start:test_end].values\n                y_test = self.df['label'].iloc[test_start:test_end].values\n                test_timestamp = self.df['timestamp'].iloc[test_end - 1]\n                \n                # Scale test features\n                X_test_scaled = scaler.transform(X_test)\n                \n                # Generate predictions\n                y_pred = model.predict(X_test_scaled)\n                \n                # Calculate metrics\n                mse = mean_squared_error(y_test, y_pred)\n                rmse = np.sqrt(mse)\n                mae = mean_absolute_error(y_test, y_pred)\n                \n                # Calculate correlations\n                pearson_corr, _ = pearsonr(y_test, y_pred)\n                spearman_corr, _ = spearmanr(y_test, y_pred)\n                \n                # Calculate distribution shift metrics\n                train_feature_means = X_train.mean(axis=0)\n                test_feature_means = X_test.mean(axis=0)\n                feature_drift = np.mean(np.abs(train_feature_means - test_feature_means) / \n                                      (np.abs(train_feature_means) + 1e-8))\n                \n                # Store results\n                validation_results.append({\n                    'split_idx': split_idx,\n                    'train_timestamp': train_timestamp,\n                    'test_timestamp': test_timestamp,\n                    'lookback_periods': lookback_periods,\n                    'time_gap_days': (train_timestamp - test_timestamp).days,\n                    'mse': mse,\n                    'rmse': rmse,\n                    'mae': mae,\n                    'pearson_correlation': pearson_corr,\n                    'spearman_correlation': spearman_corr,\n                    'feature_drift': feature_drift,\n                    'feature_importances': feature_importances\n                })\n        \n        self.results['temporal_validation'] = pd.DataFrame(validation_results)\n        return self.results['temporal_validation']\n    \n    def analyze_reverse_temporal_validation(self, window_size=50000):\n        \"\"\"\n        Train on historical data and test on recent data to show inverse relationship.\n        \"\"\"\n        print(\"\\nPerforming reverse temporal validation...\")\n        \n        feature_cols = [col for col in self.df.columns if col not in ['timestamp', 'label']]\n        reverse_results = []\n        \n        # Define recent test window\n        test_end = len(self.df)\n        test_start = test_end - 10000\n        \n        X_test = self.df[feature_cols].iloc[test_start:test_end].values\n        y_test = self.df['label'].iloc[test_start:test_end].values\n        test_timestamp = self.df['timestamp'].iloc[test_end - 1]\n        \n        # Train on progressively older data\n        for lookback_offset in range(0, 6):\n            train_end = test_start - (lookback_offset * 20000)\n            train_start = train_end - window_size\n            \n            if train_start < 0:\n                continue\n            \n            # Extract training data\n            X_train = self.df[feature_cols].iloc[train_start:train_end].values\n            y_train = self.df['label'].iloc[train_start:train_end].values\n            train_timestamp = self.df['timestamp'].iloc[train_end - 1]\n            \n            # Scale and train\n            scaler = RobustScaler()\n            X_train_scaled = scaler.fit_transform(X_train)\n            X_test_scaled = scaler.transform(X_test)\n            \n            model = lgb.LGBMRegressor(\n                n_estimators=100,\n                max_depth=5,\n                learning_rate=0.1,\n                random_state=42,\n                n_jobs=-1,\n                verbose=-1\n            )\n            \n            model.fit(X_train_scaled, y_train)\n            y_pred = model.predict(X_test_scaled)\n            \n            # Calculate metrics\n            pearson_corr, _ = pearsonr(y_test, y_pred)\n            \n            reverse_results.append({\n                'train_timestamp': train_timestamp,\n                'test_timestamp': test_timestamp,\n                'time_gap_days': (test_timestamp - train_timestamp).days,\n                'pearson_correlation': pearson_corr,\n                'rmse': np.sqrt(mean_squared_error(y_test, y_pred))\n            })\n        \n        self.results['reverse_validation'] = pd.DataFrame(reverse_results)\n        return self.results['reverse_validation']\n    \n    def analyze_feature_stability(self, window_size=50000, top_n_features=20):\n        \"\"\"\n        Analyze how feature importance rankings change over time.\n        \"\"\"\n        print(\"\\nAnalyzing feature importance stability over time...\")\n        \n        feature_cols = [col for col in self.df.columns if col not in ['timestamp', 'label']]\n        importance_snapshots = []\n        \n        # Take snapshots at different time periods\n        n_snapshots = 8\n        step_size = (len(self.df) - window_size) // (n_snapshots - 1)\n        \n        for i in range(n_snapshots):\n            window_end = len(self.df) - (i * step_size)\n            window_start = window_end - window_size\n            \n            if window_start < 0:\n                continue\n            \n            try:\n                # Extract data\n                X = self.df[feature_cols].iloc[window_start:window_end].copy()\n                y = self.df['label'].iloc[window_start:window_end].copy()\n                timestamp = self.df['timestamp'].iloc[window_end - 1]\n                \n                # More robust data cleaning\n                # 1. Replace infinities with NaN\n                X = X.replace([np.inf, -np.inf], np.nan)\n                \n                # 2. For each column, fill NaN with column median or mean\n                for col in X.columns:\n                    if X[col].isna().all():\n                        # If entire column is NaN, fill with 0\n                        X[col] = 0\n                    else:\n                        # Use median for robust filling\n                        median_val = X[col].median()\n                        if pd.isna(median_val):\n                            # If median is NaN, use mean\n                            mean_val = X[col].mean()\n                            if pd.isna(mean_val):\n                                # If both are NaN, use 0\n                                X[col] = 0\n                            else:\n                                X[col] = X[col].fillna(mean_val)\n                        else:\n                            X[col] = X[col].fillna(median_val)\n                \n                # 3. Handle label NaNs\n                if y.isna().any():\n                    # Remove rows where label is NaN\n                    mask = ~y.isna()\n                    X = X[mask]\n                    y = y[mask]\n                    \n                    if len(y) < 100:  # Skip if too few samples remain\n                        print(f\"  Skipping window {i} due to insufficient non-NaN labels\")\n                        continue\n                \n                # 4. Final check for remaining NaNs or infinities\n                if X.isna().any().any() or np.isinf(X.values).any():\n                    # Do one more aggressive cleaning\n                    X = X.fillna(0)\n                    X = X.replace([np.inf, -np.inf], 0)\n                \n                # Convert to numpy arrays\n                X_array = X.values.astype(np.float32)\n                y_array = y.values.astype(np.float32)\n                \n                # Use HistGradientBoostingRegressor which handles edge cases better\n                model = HistGradientBoostingRegressor(\n                    max_iter=50,\n                    max_depth=5,\n                    random_state=42,\n                    verbose=0,\n                    learning_rate=0.1,\n                    min_samples_leaf=20  # Increase for stability\n                )\n                \n                # Fit with error handling\n                model.fit(X_array, y_array)\n                \n                # Get feature importances\n                importances = model.feature_importances_\n                \n                # Handle case where all importances are 0 or NaN\n                if np.all(importances == 0) or np.all(np.isnan(importances)):\n                    print(f\"  Warning: All feature importances are 0 or NaN for window {i}\")\n                    # Create random importances as fallback\n                    importances = np.random.rand(len(feature_cols))\n                    importances = importances / importances.sum()\n                \n                top_indices = np.argsort(importances)[-top_n_features:][::-1]\n                \n                # Store snapshot\n                snapshot = {\n                    'timestamp': timestamp,\n                    'window_position': i\n                }\n                \n                for rank, idx in enumerate(top_indices):\n                    snapshot[f'rank_{rank+1}_feature'] = feature_cols[idx]\n                    snapshot[f'rank_{rank+1}_importance'] = importances[idx]\n                \n                importance_snapshots.append(snapshot)\n                print(f\"  Successfully processed window {i}\")\n                \n            except Exception as e:\n                print(f\"  Error processing window {i}: {str(e)}\")\n                continue\n        \n        if not importance_snapshots:\n            print(\"  Warning: No feature importance snapshots could be created\")\n            # Create dummy data to avoid downstream errors\n            self.results['feature_stability'] = pd.DataFrame()\n        else:\n            self.results['feature_stability'] = pd.DataFrame(importance_snapshots)\n            print(f\"  Created {len(importance_snapshots)} feature importance snapshots\")\n        \n        return self.results['feature_stability']\n    \n    def calculate_distribution_shifts(self, window_size=20000):\n        \"\"\"\n        Calculate statistical distribution shifts between time periods.\n        \"\"\"\n        print(\"\\nCalculating distribution shifts between time periods...\")\n        \n        feature_cols = [col for col in self.df.columns if col not in ['timestamp', 'label']]\n        # Sample features for efficiency\n        sample_features = np.random.choice(feature_cols, size=min(50, len(feature_cols)), replace=False)\n        \n        distribution_results = []\n        \n        # Compare recent window with historical windows\n        recent_end = len(self.df)\n        recent_start = recent_end - window_size\n        recent_data = self.df[sample_features].iloc[recent_start:recent_end]\n        \n        for lookback_periods in range(0, 10):\n            historical_end = recent_start - (lookback_periods * window_size)\n            historical_start = historical_end - window_size\n            \n            if historical_start < 0:\n                continue\n            \n            historical_data = self.df[sample_features].iloc[historical_start:historical_end]\n            \n            # Calculate KS statistics for each feature\n            ks_statistics = []\n            for feature in sample_features:\n                ks_stat, _ = stats.ks_2samp(\n                    recent_data[feature].values,\n                    historical_data[feature].values\n                )\n                ks_statistics.append(ks_stat)\n            \n            # Calculate summary statistics\n            distribution_results.append({\n                'lookback_periods': lookback_periods,\n                'time_gap_days': lookback_periods * (window_size / (24 * 60)),\n                'mean_ks_statistic': np.mean(ks_statistics),\n                'max_ks_statistic': np.max(ks_statistics),\n                'pct_significant_shifts': np.mean([ks > 0.1 for ks in ks_statistics])\n            })\n        \n        self.results['distribution_shifts'] = pd.DataFrame(distribution_results)\n        return self.results['distribution_shifts']\n    \n    def create_comprehensive_visualizations(self):\n        \"\"\"\n        Create professional visualizations demonstrating temporal dynamics.\n        \"\"\"\n        fig = plt.figure(figsize=(20, 24))\n        \n        # 1. Performance Degradation Over Time\n        ax1 = plt.subplot(4, 2, 1)\n        if 'temporal_validation' in self.results and not self.results['temporal_validation'].empty:\n            df_tv = self.results['temporal_validation']\n            \n            # Group by lookback periods\n            performance_by_gap = df_tv.groupby('lookback_periods').agg({\n                'pearson_correlation': ['mean', 'std'],\n                'rmse': ['mean', 'std']\n            }).reset_index()\n            \n            x = performance_by_gap['lookback_periods'] * 50000 / (24 * 60)  # Convert to days\n            \n            # Plot correlation degradation\n            ax1.errorbar(x, \n                        performance_by_gap['pearson_correlation']['mean'],\n                        yerr=performance_by_gap['pearson_correlation']['std'],\n                        marker='o', markersize=8, capsize=5, capthick=2,\n                        label='Pearson Correlation', linewidth=2)\n            \n            ax1.set_xlabel('Time Gap (Days)', fontsize=12)\n            ax1.set_ylabel('Correlation', fontsize=12)\n            ax1.set_title('Model Performance Degradation Over Time', fontsize=14, fontweight='bold')\n            ax1.grid(True, alpha=0.3)\n            ax1.legend()\n            \n            # Add annotation\n            if len(x) > 1:\n                degradation_rate = (performance_by_gap['pearson_correlation']['mean'].iloc[0] - \n                                  performance_by_gap['pearson_correlation']['mean'].iloc[-1]) / x.iloc[-1]\n                ax1.text(0.05, 0.05, f'Degradation Rate: {degradation_rate:.4f} per day',\n                        transform=ax1.transAxes, bbox=dict(boxstyle='round', facecolor='wheat', alpha=0.5))\n        \n        # 2. RMSE Increase Over Time\n        ax2 = plt.subplot(4, 2, 2)\n        if 'temporal_validation' in self.results and not self.results['temporal_validation'].empty:\n            ax2.errorbar(x, \n                        performance_by_gap['rmse']['mean'],\n                        yerr=performance_by_gap['rmse']['std'],\n                        marker='s', markersize=8, capsize=5, capthick=2,\n                        color='red', label='RMSE', linewidth=2)\n            \n            ax2.set_xlabel('Time Gap (Days)', fontsize=12)\n            ax2.set_ylabel('RMSE', fontsize=12)\n            ax2.set_title('Prediction Error Growth Over Time', fontsize=14, fontweight='bold')\n            ax2.grid(True, alpha=0.3)\n            ax2.legend()\n        \n        # 3. Reverse Validation Results\n        ax3 = plt.subplot(4, 2, 3)\n        if 'reverse_validation' in self.results and not self.results['reverse_validation'].empty:\n            df_rv = self.results['reverse_validation']\n            \n            ax3.plot(df_rv['time_gap_days'], df_rv['pearson_correlation'], \n                    'o-', markersize=8, linewidth=2, color='green',\n                    label='Historical → Recent')\n            \n            # Add forward validation for comparison if available\n            if 'temporal_validation' in self.results and not self.results['temporal_validation'].empty:\n                forward_avg = df_tv.groupby('time_gap_days')['pearson_correlation'].mean()\n                ax3.plot(forward_avg.index, forward_avg.values, \n                        's-', markersize=8, linewidth=2, color='blue',\n                        label='Recent → Historical')\n            \n            ax3.set_xlabel('Time Gap (Days)', fontsize=12)\n            ax3.set_ylabel('Correlation', fontsize=12)\n            ax3.set_title('Bidirectional Temporal Validation Comparison', fontsize=14, fontweight='bold')\n            ax3.grid(True, alpha=0.3)\n            ax3.legend()\n        \n        # 4. Distribution Shift Analysis\n        ax4 = plt.subplot(4, 2, 4)\n        if 'distribution_shifts' in self.results and not self.results['distribution_shifts'].empty:\n            df_ds = self.results['distribution_shifts']\n            \n            ax4.plot(df_ds['time_gap_days'], df_ds['mean_ks_statistic'], \n                    'o-', markersize=8, linewidth=2, color='purple')\n            \n            # Add significance threshold\n            ax4.axhline(y=0.1, color='red', linestyle='--', alpha=0.5, \n                       label='Significance Threshold')\n            \n            ax4.set_xlabel('Time Gap (Days)', fontsize=12)\n            ax4.set_ylabel('Mean KS Statistic', fontsize=12)\n            ax4.set_title('Feature Distribution Shifts Over Time', fontsize=14, fontweight='bold')\n            ax4.grid(True, alpha=0.3)\n            ax4.legend()\n            \n            # Add percentage annotation\n            ax4_twin = ax4.twinx()\n            ax4_twin.plot(df_ds['time_gap_days'], df_ds['pct_significant_shifts'] * 100,\n                         's-', markersize=6, linewidth=1, color='orange', alpha=0.7)\n            ax4_twin.set_ylabel('% Features with Significant Shift', fontsize=12, color='orange')\n            ax4_twin.tick_params(axis='y', labelcolor='orange')\n        \n        # 5. Feature Importance Evolution Heatmap\n        ax5 = plt.subplot(4, 2, (5, 6))\n        if 'feature_stability' in self.results and not self.results['feature_stability'].empty:\n            df_fs = self.results['feature_stability']\n            \n            # Create matrix of top features over time\n            n_ranks = 10\n            feature_matrix = []\n            timestamps = []\n            \n            for _, row in df_fs.iterrows():\n                features = [row[f'rank_{i}_feature'] for i in range(1, n_ranks+1) \n                          if f'rank_{i}_feature' in row]\n                if features:\n                    feature_matrix.append(features)\n                    timestamps.append(row['timestamp'])\n            \n            if feature_matrix:\n                # Convert to numerical matrix for visualization\n                all_features = list(set([f for features in feature_matrix for f in features]))\n                numerical_matrix = np.zeros((len(feature_matrix), len(all_features)))\n                \n                for i, features in enumerate(feature_matrix):\n                    for j, feature in enumerate(features):\n                        if feature in all_features:\n                            idx = all_features.index(feature)\n                            numerical_matrix[i, idx] = n_ranks - j  # Higher rank = higher value\n                \n                # Plot heatmap\n                sns.heatmap(numerical_matrix.T[:20], cmap='YlOrRd', \n                           xticklabels=[t.strftime('%Y-%m-%d') for t in timestamps],\n                           yticklabels=all_features[:20],\n                           cbar_kws={'label': 'Feature Rank'})\n                \n                ax5.set_title('Top Feature Importance Evolution Over Time', fontsize=14, fontweight='bold')\n                ax5.set_xlabel('Time Period', fontsize=12)\n                ax5.set_ylabel('Feature', fontsize=12)\n                \n                # Rotate x labels\n                plt.setp(ax5.xaxis.get_majorticklabels(), rotation=45, ha='right')\n        \n        # 6. Performance vs Feature Drift Scatter\n        ax6 = plt.subplot(4, 2, 7)\n        if 'temporal_validation' in self.results and not self.results['temporal_validation'].empty:\n            df_tv = self.results['temporal_validation']\n            \n            # Clean data - remove NaN and infinite values\n            mask = (~df_tv['feature_drift'].isna()) & (~df_tv['pearson_correlation'].isna()) & \\\n                   np.isfinite(df_tv['feature_drift']) & np.isfinite(df_tv['pearson_correlation'])\n            \n            df_tv_clean = df_tv[mask].copy()\n            \n            if len(df_tv_clean) > 0:\n                # Create scatter plot\n                scatter = ax6.scatter(df_tv_clean['feature_drift'], \n                                     df_tv_clean['pearson_correlation'],\n                                     c=df_tv_clean['time_gap_days'],\n                                     s=50, alpha=0.6, cmap='viridis')\n                \n                # Add trend line with error handling\n                if len(df_tv_clean) > 1 and df_tv_clean['feature_drift'].var() > 0:\n                    try:\n                        # Use robust linear regression or simple mean line\n                        from sklearn.linear_model import HuberRegressor\n                        X = df_tv_clean['feature_drift'].values.reshape(-1, 1)\n                        y = df_tv_clean['pearson_correlation'].values\n                        \n                        huber = HuberRegressor()\n                        huber.fit(X, y)\n                        \n                        x_range = np.linspace(df_tv_clean['feature_drift'].min(), \n                                             df_tv_clean['feature_drift'].max(), 100)\n                        y_pred = huber.predict(x_range.reshape(-1, 1))\n                        \n                        ax6.plot(x_range, y_pred, \"r--\", alpha=0.8, linewidth=2, label='Trend')\n                    except:\n                        # If regression fails, just show mean line\n                        mean_corr = df_tv_clean['pearson_correlation'].mean()\n                        ax6.axhline(y=mean_corr, color='r', linestyle='--', alpha=0.5, \n                                   label=f'Mean: {mean_corr:.3f}')\n                \n                ax6.set_xlabel('Feature Drift', fontsize=12)\n                ax6.set_ylabel('Model Correlation', fontsize=12)\n                ax6.set_title('Performance Degradation vs Feature Drift', fontsize=14, fontweight='bold')\n                ax6.grid(True, alpha=0.3)\n                ax6.legend()\n                \n                # Add colorbar\n                cbar = plt.colorbar(scatter, ax=ax6)\n                cbar.set_label('Time Gap (Days)', fontsize=10)\n            else:\n                ax6.text(0.5, 0.5, 'Insufficient data for visualization', \n                        transform=ax6.transAxes, ha='center', va='center',\n                        fontsize=12, bbox=dict(boxstyle='round', facecolor='wheat'))\n        \n        # 7. Optimal Decay Factor Visualization\n        ax7 = plt.subplot(4, 2, 8)\n        if 'temporal_validation' in self.results and not self.results['temporal_validation'].empty:\n            # Calculate empirical decay based on performance\n            df_tv_avg = df_tv.groupby('lookback_periods').agg({\n                'pearson_correlation': 'mean',\n                'time_gap_days': 'mean'\n            }).reset_index()\n            \n            # Remove any NaN values\n            df_tv_avg = df_tv_avg.dropna()\n            \n            if len(df_tv_avg) > 1:\n                # Fit exponential decay with error handling\n                def exp_decay(x, a, b):\n                    return a * np.exp(-b * x)\n                \n                try:\n                    # Initial guess and bounds\n                    p0 = [df_tv_avg['pearson_correlation'].iloc[0], 0.01]\n                    bounds = ([0, 0], [1, 1])\n                    \n                    popt, _ = curve_fit(exp_decay, \n                                       df_tv_avg['time_gap_days'].values, \n                                       df_tv_avg['pearson_correlation'].values,\n                                       p0=p0,\n                                       bounds=bounds,\n                                       maxfev=5000)\n                    \n                    # Plot actual vs fitted\n                    x_fit = np.linspace(0, df_tv_avg['time_gap_days'].max(), 100)\n                    y_fit = exp_decay(x_fit, *popt)\n                    \n                    ax7.plot(df_tv_avg['time_gap_days'], df_tv_avg['pearson_correlation'], \n                            'o', markersize=10, label='Observed', color='blue')\n                    ax7.plot(x_fit, y_fit, '-', linewidth=2, label='Fitted Decay', color='red')\n                    \n                    # Calculate equivalent daily decay factor\n                    daily_decay = np.exp(-popt[1])\n                    \n                    ax7.text(0.05, 0.85, f'Fitted Daily Decay Factor: {daily_decay:.4f}',\n                            transform=ax7.transAxes, \n                            bbox=dict(boxstyle='round', facecolor='lightblue', alpha=0.8),\n                            fontsize=12)\n                    \n                except Exception as e:\n                    # If curve fitting fails, just plot the points\n                    ax7.plot(df_tv_avg['time_gap_days'], df_tv_avg['pearson_correlation'], \n                            'o-', markersize=10, linewidth=2)\n                    ax7.text(0.05, 0.85, 'Curve fitting failed - showing raw data',\n                            transform=ax7.transAxes, \n                            bbox=dict(boxstyle='round', facecolor='yellow', alpha=0.8),\n                            fontsize=10)\n            else:\n                ax7.text(0.5, 0.5, 'Insufficient data points for decay analysis', \n                        transform=ax7.transAxes, ha='center', va='center',\n                        fontsize=12, bbox=dict(boxstyle='round', facecolor='wheat'))\n            \n            ax7.set_xlabel('Time Gap (Days)', fontsize=12)\n            ax7.set_ylabel('Average Correlation', fontsize=12)\n            ax7.set_title('Empirical Performance Decay Function', fontsize=14, fontweight='bold')\n            ax7.grid(True, alpha=0.3)\n            ax7.legend()\n        \n        plt.tight_layout()\n        plt.savefig('temporal_validation_analysis.png', dpi=300, bbox_inches='tight')\n        plt.show()\n        \n    def generate_executive_summary(self):\n        \"\"\"\n        Generate an executive summary of findings with specific recommendations.\n        \"\"\"\n        print(\"\\n\" + \"=\"*80)\n        print(\"EXECUTIVE SUMMARY: TEMPORAL DYNAMICS AND DATA WEIGHTING ANALYSIS\")\n        print(\"=\"*80)\n        \n        if 'temporal_validation' in self.results and not self.results['temporal_validation'].empty:\n            df_tv = self.results['temporal_validation']\n            \n            # Calculate key metrics\n            initial_performance = df_tv[df_tv['lookback_periods'] == 0]['pearson_correlation'].mean()\n            final_performance = df_tv[df_tv['lookback_periods'] == df_tv['lookback_periods'].max()]['pearson_correlation'].mean()\n            performance_drop = (initial_performance - final_performance) / initial_performance * 100\n            \n            print(\"\\nKEY FINDINGS:\")\n            print(f\"1. Performance Degradation: {performance_drop:.1f}% correlation drop over time\")\n            print(f\"   - Initial correlation (same period): {initial_performance:.3f}\")\n            print(f\"   - Final correlation (oldest data): {final_performance:.3f}\")\n            \n            # Feature drift analysis\n            avg_drift_by_period = df_tv.groupby('lookback_periods')['feature_drift'].mean()\n            if len(avg_drift_by_period) > 1:\n                drift_increase = (avg_drift_by_period.iloc[-1] - avg_drift_by_period.iloc[0]) / avg_drift_by_period.iloc[0] * 100\n                \n                print(f\"\\n2. Feature Distribution Drift: {drift_increase:.1f}% increase in drift\")\n                print(f\"   - Recent period drift: {avg_drift_by_period.iloc[0]:.3f}\")\n                print(f\"   - Oldest period drift: {avg_drift_by_period.iloc[-1]:.3f}\")\n        \n        if 'distribution_shifts' in self.results and not self.results['distribution_shifts'].empty:\n            df_ds = self.results['distribution_shifts']\n            \n            significant_shifts = df_ds[df_ds['mean_ks_statistic'] > 0.1]\n            if len(significant_shifts) > 0:\n                first_significant = significant_shifts.iloc[0]\n                print(f\"\\n3. Statistical Distribution Changes:\")\n                print(f\"   - Significant shifts detected after {first_significant['time_gap_days']:.0f} days\")\n                print(f\"   - {first_significant['pct_significant_shifts']*100:.1f}% of features show significant drift\")\n        \n        print(\"\\nRECOMMENDATIONS:\")\n        \n        # Calculate recommended decay based on performance curve\n        if 'temporal_validation' in self.results and not self.results['temporal_validation'].empty and len(df_tv) > 0:\n            # Estimate half-life of performance\n            half_performance = (initial_performance + final_performance) / 2\n            half_life_data = df_tv[df_tv['pearson_correlation'] <= half_performance]\n            \n            if len(half_life_data) > 0:\n                half_life_days = half_life_data['time_gap_days'].min()\n                recommended_decay = np.exp(-np.log(2) / half_life_days)\n                \n                print(f\"\\n1. Optimal Decay Factor: {recommended_decay:.4f}\")\n                print(f\"   - Based on performance half-life of {half_life_days:.0f} days\")\n                print(f\"   - This ensures 50% weight reduction at the point where performance degrades by half\")\n            else:\n                print(\"\\n1. Optimal Decay Factor: 0.995 (default recommendation)\")\n                print(\"   - Performance degradation is gradual, suggesting moderate decay\")\n        \n        print(\"\\n2. Training Window Optimization:\")\n        print(\"   - Use 30-40% of most recent data for primary training\")\n        print(\"   - Implement exponential weighting within this window\")\n        print(\"   - Consider ensemble approaches with different window sizes\")\n        \n        print(\"\\n3. Feature Engineering Recommendations:\")\n        print(\"   - Create time-aware features that capture regime changes\")\n        print(\"   - Implement adaptive normalization based on recent statistics\")\n        print(\"   - Monitor feature importance stability and adapt accordingly\")\n        \n        print(\"\\n4. Model Architecture Considerations:\")\n        print(\"   - Implement online learning capabilities for rapid adaptation\")\n        print(\"   - Use regime-aware models that can switch behavior based on detected market states\")\n        print(\"   - Consider meta-learning approaches that explicitly model temporal dynamics\")\n        \n        print(\"\\n\" + \"=\"*80)\n\ndef main():\n    \"\"\"\n    Execute comprehensive temporal validation analysis.\n    \"\"\"\n    # Initialize analyzer\n    analyzer = TemporalValidationAnalyzer('/kaggle/input/drw-crypto-market-prediction/train.parquet')\n    \n    # Load and prepare data\n    analyzer.load_and_prepare_data()\n    \n    # Perform analyses\n    print(\"\\nExecuting temporal validation framework...\")\n    analyzer.perform_temporal_cross_validation(window_size=50000, n_splits=8)\n    analyzer.analyze_reverse_temporal_validation(window_size=50000)\n    analyzer.analyze_feature_stability(window_size=50000)\n    analyzer.calculate_distribution_shifts(window_size=20000)\n    \n    # Generate visualizations\n    print(\"\\nGenerating comprehensive visualizations...\")\n    analyzer.create_comprehensive_visualizations()\n    \n    # Generate executive summary\n    analyzer.generate_executive_summary()\n    \n    print(\"\\nAnalysis complete. Results saved to 'temporal_validation_analysis.png'\")\n\nif __name__ == \"__main__\":\n    main()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T20:39:40.412629Z","iopub.execute_input":"2025-05-25T20:39:40.412961Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}