{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":96164,"databundleVersionId":12993472,"sourceType":"competition"}],"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport os\n\n# Input data files are available in the read-only \"../input/\" directory\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# Install required packages\nprint(\"Installing required packages...\")\n!pip install koolbox scikit-learn==1.5.2 prophet --quiet\n\n# Import all required libraries\nimport pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nfrom prophet import Prophet\nimport warnings\nwarnings.filterwarnings('ignore')\n\nfrom sklearn.model_selection import KFold, TimeSeriesSplit\nfrom sklearn.linear_model import Ridge\nfrom lightgbm import LGBMRegressor\nfrom scipy.stats import pearsonr, norm, gaussian_kde\nfrom xgboost import XGBRegressor\nfrom sklearn.base import clone\nfrom koolbox import Trainer\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport pandas as pd\nimport numpy as np\nimport warnings\nimport optuna\nimport joblib\nimport gc\nfrom scipy import stats\nimport pickle\nfrom dataclasses import dataclass\nfrom typing import Dict, Tuple, List, Optional\nfrom collections import defaultdict\nfrom sklearn.mixture import GaussianMixture\nfrom sklearn.preprocessing import RobustScaler\nfrom sklearn.isotonic import IsotonicRegression\n\nwarnings.filterwarnings(\"ignore\")\n\nclass CFG:\n    train_path = \"/kaggle/input/drw-crypto-market-prediction/train.parquet\"\n    test_path = \"/kaggle/input/drw-crypto-market-prediction/test.parquet\"\n    sample_sub_path = \"/kaggle/input/drw-crypto-market-prediction/sample_submission.csv\"\n\n    target = \"label\"\n    n_folds = 10  # Increased for better stability analysis\n    n_stability_folds = 5  # Additional folds for stability testing\n    seed = 42\n\n@dataclass\nclass ProphetGuidedProfile:\n    \"\"\"Stores Prophet-guided compression parameters that can be applied without timestamps\"\"\"\n    median: float\n    mad: float\n    percentiles: Dict[int, float]  # 1, 5, 10, 25, 75, 90, 95, 99\n    kde: Optional[gaussian_kde]  # Kernel density estimator\n    outlier_bounds: Tuple[float, float]  # Lower and upper bounds for outliers\n    compression_map: IsotonicRegression  # Maps values to compression strengths\n    prophet_outlier_percentiles: List[float]  # Percentiles of Prophet-identified outliers\n    value_compression_dict: Dict[float, float]  # Discretized value->compression mapping\n\n@dataclass\nclass StabilityMetrics:\n    \"\"\"Stores stability metrics for a strategy\"\"\"\n    mean_cv_score: float\n    std_cv_score: float\n    stability_score: float  # mean - 2*std (higher is better)\n    fold_scores: List[float]\n    inter_fold_correlation: float\n\nclass ProphetGuidedCompressor:\n    \"\"\"Hybrid compressor using Prophet to guide distribution-based compression\"\"\"\n    \n    def __init__(self):\n        self.feature_profiles: Dict[str, ProphetGuidedProfile] = {}\n        self.label_profile: ProphetGuidedProfile = None\n        self.scalers: Dict[str, RobustScaler] = {}\n        \n    def prophet_guided_analysis(self, df: pd.DataFrame, feature_col: str, \n                               timestamp_col: str = '__index_level_0__') -> Dict[str, any]:\n        \"\"\"Use Prophet to identify outliers and guide compression strength learning\"\"\"\n        \n        print(f\"  Prophet-guided analysis for {feature_col}...\")\n        \n        # Prepare data for Prophet\n        prophet_df = pd.DataFrame({\n            'ds': df[timestamp_col],\n            'y': df[feature_col]\n        })\n        \n        # Remove NaN values\n        prophet_df = prophet_df[np.isfinite(prophet_df['y'])]\n        \n        if len(prophet_df) < 100:\n            return None\n        \n        try:\n            # Resample to hourly for efficiency\n            prophet_hourly = prophet_df.set_index('ds').resample('1H').mean().reset_index()\n            prophet_hourly = prophet_hourly.dropna()\n            \n            # Fit Prophet model\n            model = Prophet(\n                changepoint_prior_scale=0.05,\n                interval_width=0.95,\n                yearly_seasonality=False,\n                weekly_seasonality=True,\n                daily_seasonality=True,\n                stan_backend='cmdstanpy'\n            )\n            \n            model.fit(prophet_hourly)\n            \n            # Generate predictions\n            forecast = model.predict(prophet_hourly)\n            \n            # Calculate residuals\n            residuals = prophet_hourly['y'].values - forecast['yhat'].values\n            residual_std = np.std(residuals)\n            \n            # Create value->outlier_score mapping for original data\n            outlier_scores = {}\n            \n            for idx, row in prophet_df.iterrows():\n                # Find nearest forecast time\n                time_diffs = np.abs(forecast['ds'] - row['ds'])\n                nearest_idx = np.argmin(time_diffs)\n                \n                # Calculate outlier score based on Prophet prediction\n                yhat = forecast.iloc[nearest_idx]['yhat']\n                yhat_lower = forecast.iloc[nearest_idx]['yhat_lower']\n                yhat_upper = forecast.iloc[nearest_idx]['yhat_upper']\n                \n                # Outlier score: 0 = perfectly normal, 1 = extreme outlier\n                if row['y'] < yhat_lower:\n                    score = min(1.0, (yhat_lower - row['y']) / (3 * residual_std))\n                elif row['y'] > yhat_upper:\n                    score = min(1.0, (row['y'] - yhat_upper) / (3 * residual_std))\n                else:\n                    score = 0\n                \n                outlier_scores[row['y']] = score\n            \n            return {\n                'outlier_scores': outlier_scores,\n                'residual_std': residual_std,\n                'prophet_model': model,\n                'forecast': forecast\n            }\n            \n        except Exception as e:\n            print(f\"    Prophet failed: {str(e)}\")\n            return None\n    \n    def create_compression_mapping(self, values: np.ndarray, \n                                 prophet_results: Optional[Dict] = None) -> IsotonicRegression:\n        \"\"\"Create a mapping from values to compression strengths\"\"\"\n        \n        # Remove NaN values\n        clean_values = values[~np.isnan(values)]\n        \n        if len(clean_values) < 10:\n            # Return identity mapping\n            iso_reg = IsotonicRegression(out_of_bounds='clip')\n            iso_reg.fit([0, 1], [0, 0])\n            return iso_reg\n        \n        # Calculate base compression from distribution\n        percentiles = np.percentile(clean_values, [1, 5, 10, 25, 50, 75, 90, 95, 99])\n        \n        # Base compression strengths\n        compression_strengths = []\n        value_points = []\n        \n        for val in clean_values:\n            # Find position in distribution\n            if val <= percentiles[0]:  # Below 1st percentile\n                base_strength = 0.8\n            elif val <= percentiles[1]:  # 1-5th percentile\n                base_strength = 0.6\n            elif val <= percentiles[2]:  # 5-10th percentile\n                base_strength = 0.4\n            elif val <= percentiles[3]:  # 10-25th percentile\n                base_strength = 0.25\n            elif val >= percentiles[8]:  # Above 99th percentile\n                base_strength = 0.8\n            elif val >= percentiles[7]:  # 95-99th percentile\n                base_strength = 0.6\n            elif val >= percentiles[6]:  # 90-95th percentile\n                base_strength = 0.4\n            elif val >= percentiles[5]:  # 75-90th percentile\n                base_strength = 0.25\n            else:  # Central 50%\n                base_strength = 0.1\n            \n            # Adjust by Prophet outlier score if available\n            if prophet_results and 'outlier_scores' in prophet_results:\n                # Find closest value in outlier_scores\n                outlier_score = 0\n                for out_val, score in prophet_results['outlier_scores'].items():\n                    if abs(out_val - val) < 1e-6:\n                        outlier_score = score\n                        break\n                \n                # Combine base and Prophet-guided compression\n                # Higher outlier score -> more compression\n                final_strength = base_strength * 0.6 + outlier_score * 0.4\n            else:\n                final_strength = base_strength\n            \n            compression_strengths.append(final_strength)\n            value_points.append(val)\n        \n        # Fit isotonic regression for smooth monotonic mapping\n        sorted_indices = np.argsort(value_points)\n        sorted_values = np.array(value_points)[sorted_indices]\n        sorted_strengths = np.array(compression_strengths)[sorted_indices]\n        \n        # Create isotonic regression model\n        iso_reg = IsotonicRegression(out_of_bounds='clip')\n        iso_reg.fit(sorted_values, sorted_strengths)\n        \n        return iso_reg\n    \n    def fit(self, X: pd.DataFrame, y: pd.Series = None, \n            base_compression: float = 0.3,\n            label_compression: float = 0.0):\n        \"\"\"Learn compression using Prophet guidance where available\"\"\"\n        \n        print(\"Learning Prophet-guided compression parameters...\")\n        \n        # Check if we have timestamp\n        has_timestamp = '__index_level_0__' in X.columns\n        \n        if not has_timestamp:\n            # Add synthetic timestamp for Prophet\n            X = X.copy()\n            X['__index_level_0__'] = pd.date_range('2023-03-01', periods=len(X), freq='T')\n        \n        # Analyze each feature\n        for col in X.columns:\n            if col == '__index_level_0__':\n                continue\n                \n            # Get feature data\n            feature_data = X[col].values\n            \n            # Try Prophet analysis\n            prophet_results = None\n            if has_timestamp or True:  # Always try Prophet with synthetic timestamps\n                temp_df = pd.DataFrame({\n                    '__index_level_0__': X['__index_level_0__'],\n                    col: X[col]\n                })\n                prophet_results = self.prophet_guided_analysis(temp_df, col)\n            \n            # Calculate distribution statistics\n            clean_data = feature_data[~np.isnan(feature_data)]\n            median = np.median(clean_data)\n            mad = np.median(np.abs(clean_data - median))\n            if mad == 0:\n                mad = np.std(clean_data)\n            \n            # Calculate percentiles\n            percentiles = {}\n            for p in [1, 5, 10, 25, 50, 75, 90, 95, 99]:\n                percentiles[p] = np.percentile(clean_data, p)\n            \n            # Fit KDE\n            try:\n                kde = gaussian_kde(clean_data, bw_method='scott')\n            except:\n                kde = None\n            \n            # Define outlier bounds\n            iqr = percentiles[75] - percentiles[25]\n            outlier_bounds = (\n                percentiles[25] - 3 * iqr,\n                percentiles[75] + 3 * iqr\n            )\n            \n            # Create compression mapping\n            compression_map = self.create_compression_mapping(feature_data, prophet_results)\n            \n            # Extract Prophet outlier percentiles if available\n            prophet_outlier_percentiles = []\n            if prophet_results and 'outlier_scores' in prophet_results:\n                outlier_values = [v for v, s in prophet_results['outlier_scores'].items() if s > 0.5]\n                if outlier_values:\n                    prophet_outlier_percentiles = [\n                        stats.percentileofscore(clean_data, v) for v in outlier_values\n                    ]\n            \n            # Create discretized mapping for fast lookup\n            value_range = np.linspace(clean_data.min(), clean_data.max(), 1000)\n            compression_strengths = compression_map.predict(value_range)\n            value_compression_dict = dict(zip(value_range, compression_strengths))\n            \n            # Store profile\n            self.feature_profiles[col] = ProphetGuidedProfile(\n                median=median,\n                mad=mad,\n                percentiles=percentiles,\n                kde=kde,\n                outlier_bounds=outlier_bounds,\n                compression_map=compression_map,\n                prophet_outlier_percentiles=prophet_outlier_percentiles,\n                value_compression_dict=value_compression_dict\n            )\n            \n            # Print summary\n            if prophet_results:\n                n_outliers = len([s for s in prophet_results['outlier_scores'].values() if s > 0.5])\n                print(f\"    {col}: Prophet found {n_outliers} outliers, \"\n                      f\"compression range: {compression_strengths.min():.3f}-{compression_strengths.max():.3f}\")\n            else:\n                print(f\"    {col}: Distribution-based compression, \"\n                      f\"range: {compression_strengths.min():.3f}-{compression_strengths.max():.3f}\")\n        \n        # Analyze label if provided\n        if y is not None and label_compression > 0:\n            print(\"  Analyzing label distribution...\")\n            # For labels, use only distribution-based compression (no Prophet)\n            clean_labels = y.values[~np.isnan(y.values)]\n            \n            median = np.median(clean_labels)\n            mad = np.median(np.abs(clean_labels - median))\n            if mad == 0:\n                mad = np.std(clean_labels)\n            \n            percentiles = {}\n            for p in [1, 5, 10, 25, 50, 75, 90, 95, 99]:\n                percentiles[p] = np.percentile(clean_labels, p)\n            \n            # Simple compression mapping for labels\n            iso_reg = IsotonicRegression(out_of_bounds='clip')\n            iso_reg.fit([percentiles[1], percentiles[50], percentiles[99]], \n                       [0.1, 0.05, 0.1])  # Minimal compression for labels\n            \n            self.label_profile = ProphetGuidedProfile(\n                median=median,\n                mad=mad,\n                percentiles=percentiles,\n                kde=None,\n                outlier_bounds=(percentiles[1], percentiles[99]),\n                compression_map=iso_reg,\n                prophet_outlier_percentiles=[],\n                value_compression_dict={}\n            )\n    \n    def transform(self, X: pd.DataFrame, y: pd.Series = None,\n                 base_compression: float = 0.3) -> Tuple[pd.DataFrame, pd.Series]:\n        \"\"\"Apply learned compression without requiring timestamps\"\"\"\n        \n        X_compressed = X.copy()\n        \n        # Remove timestamp column if present\n        if '__index_level_0__' in X_compressed.columns:\n            X_compressed = X_compressed.drop('__index_level_0__', axis=1)\n        \n        # Compress each feature\n        for col in X_compressed.columns:\n            if col in self.feature_profiles:\n                profile = self.feature_profiles[col]\n                X_compressed[col] = self._compress_feature(\n                    X_compressed[col].values,\n                    profile,\n                    base_compression\n                )\n        \n        # Compress labels if needed\n        y_compressed = y\n        if y is not None and self.label_profile is not None:\n            y_compressed = pd.Series(\n                self._compress_feature(\n                    y.values,\n                    self.label_profile,\n                    base_compression * 0.3  # Much more conservative for labels\n                ),\n                index=y.index\n            )\n        \n        return X_compressed, y_compressed\n    \n    def _compress_feature(self, data: np.ndarray, profile: ProphetGuidedProfile,\n                         base_compression: float) -> np.ndarray:\n        \"\"\"Apply compression using learned mapping\"\"\"\n        \n        # Handle NaN values\n        nan_mask = np.isnan(data)\n        result = data.copy()\n        \n        if profile.mad == 0:\n            return data\n        \n        # Normalize data\n        normalized = (data - profile.median) / (profile.mad * 6)\n        \n        # Apply compression based on learned mapping\n        for i in range(len(data)):\n            if nan_mask[i]:\n                continue\n            \n            # Get compression strength from learned mapping\n            compression_strength = profile.compression_map.predict([data[i]])[0]\n            \n            # Scale by base compression\n            final_compression = base_compression * compression_strength\n            \n            # Apply smooth compression (tanh)\n            compressed_normalized = np.tanh(normalized[i] * (1 - final_compression))\n            \n            # Scale back\n            result[i] = compressed_normalized * (profile.mad * 6) + profile.median\n        \n        return result\n    \n    def save(self, filepath: str):\n        \"\"\"Save compression profiles\"\"\"\n        # Can't pickle all objects, so save essential parameters\n        save_data = {\n            'feature_profiles': {},\n            'label_profile': None,\n            'scalers': self.scalers\n        }\n        \n        for feature, profile in self.feature_profiles.items():\n            # Save isotonic regression parameters\n            X_min = profile.compression_map.X_min_ if hasattr(profile.compression_map, 'X_min_') else 0\n            X_max = profile.compression_map.X_max_ if hasattr(profile.compression_map, 'X_max_') else 1\n            \n            save_data['feature_profiles'][feature] = {\n                'median': profile.median,\n                'mad': profile.mad,\n                'percentiles': profile.percentiles,\n                'outlier_bounds': profile.outlier_bounds,\n                'prophet_outlier_percentiles': profile.prophet_outlier_percentiles,\n                'value_compression_dict': profile.value_compression_dict,\n                'iso_X_min': X_min,\n                'iso_X_max': X_max\n            }\n        \n        if self.label_profile:\n            save_data['label_profile'] = {\n                'median': self.label_profile.median,\n                'mad': self.label_profile.mad,\n                'percentiles': self.label_profile.percentiles,\n                'outlier_bounds': self.label_profile.outlier_bounds\n            }\n        \n        with open(filepath, 'wb') as f:\n            pickle.dump(save_data, f)\n\ndef calculate_stability_metrics(scores_dict: Dict[str, List[float]]) -> StabilityMetrics:\n    \"\"\"Calculate stability metrics across folds\"\"\"\n    all_scores = []\n    for model_scores in scores_dict.values():\n        all_scores.extend(model_scores)\n    \n    mean_score = np.mean(all_scores)\n    std_score = np.std(all_scores)\n    \n    # Calculate inter-fold correlation\n    fold_predictions = defaultdict(list)\n    for model, scores in scores_dict.items():\n        for i, score in enumerate(scores):\n            fold_predictions[i].append(score)\n    \n    # Average correlation between fold results\n    correlations = []\n    folds = list(fold_predictions.keys())\n    for i in range(len(folds)):\n        for j in range(i+1, len(folds)):\n            corr = np.corrcoef(fold_predictions[folds[i]], fold_predictions[folds[j]])[0, 1]\n            if not np.isnan(corr):\n                correlations.append(corr)\n    \n    inter_fold_corr = np.mean(correlations) if correlations else 0\n    \n    return StabilityMetrics(\n        mean_cv_score=mean_score,\n        std_cv_score=std_score,\n        stability_score=mean_score - 2 * std_score,  # Conservative stability metric\n        fold_scores=all_scores,\n        inter_fold_correlation=inter_fold_corr\n    )\n\ndef reduce_mem_usage(dataframe, dataset):    \n    print('Reducing memory usage for:', dataset)\n    initial_mem_usage = dataframe.memory_usage().sum() / 1024**2\n    \n    for col in dataframe.columns:\n        col_type = dataframe[col].dtype\n\n        c_min = dataframe[col].min()\n        c_max = dataframe[col].max()\n        if str(col_type)[:3] == 'int':\n            if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                dataframe[col] = dataframe[col].astype(np.int8)\n            elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                dataframe[col] = dataframe[col].astype(np.int16)\n            elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                dataframe[col] = dataframe[col].astype(np.int32)\n            elif c_min > np.iinfo(np.int64).min and c_max < np.iinfo(np.int64).max:\n                dataframe[col] = dataframe[col].astype(np.int64)\n        else:\n            if c_min > np.finfo(np.float16).min and c_max < np.finfo(np.float16).max:\n                dataframe[col] = dataframe[col].astype(np.float16)\n            elif c_min > np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n                dataframe[col] = dataframe[col].astype(np.float32)\n            else:\n                dataframe[col] = dataframe[col].astype(np.float64)\n\n    final_mem_usage = dataframe.memory_usage().sum() / 1024**2\n    print('--- Memory usage before: {:.2f} MB'.format(initial_mem_usage))\n    print('--- Memory usage after: {:.2f} MB'.format(final_mem_usage))\n    print('--- Decreased memory usage by {:.1f}%\\n'.format(100 * (initial_mem_usage - final_mem_usage) / initial_mem_usage))\n\n    return dataframe\n\ndef _pearsonr(y_true, y_pred):\n    return pearsonr(y_true, y_pred)[0]\n\n# Define model parameters\nlgbm_params = {\n    \"boosting_type\": \"gbdt\",\n    \"colsample_bytree\": 0.5625888953382505,\n    \"learning_rate\": 0.029312951475451557,\n    \"min_child_samples\": 63,\n    \"min_child_weight\": 0.11456572852335424,\n    \"n_estimators\": 126,\n    \"n_jobs\": -1,\n    \"num_leaves\": 37,\n    \"random_state\": 42,\n    \"reg_alpha\": 85.2476527854083,\n    \"reg_lambda\": 99.38305361388907,\n    \"subsample\": 0.450669817684892,\n    \"verbose\": -1\n}\n\nlgbm_goss_params = {\n    \"boosting_type\": \"goss\",\n    \"colsample_bytree\": 0.34695458228489784,\n    \"learning_rate\": 0.031023014900595287,\n    \"min_child_samples\": 30,\n    \"min_child_weight\": 0.4727729225033618,\n    \"n_estimators\": 220,\n    \"n_jobs\": -1,\n    \"num_leaves\": 58,\n    \"random_state\": 42,\n    \"reg_alpha\": 38.665994901468224,\n    \"reg_lambda\": 92.76991677464294,\n    \"subsample\": 0.4810891284493255,\n    \"verbose\": -1\n}\n\nxgb_params = {\n    \"colsample_bylevel\": 0.4778015829774066,\n    \"colsample_bynode\": 0.362764358742407,\n    \"colsample_bytree\": 0.7107423488010493,\n    \"gamma\": 1.7094857725240398,\n    \"learning_rate\": 0.02213323588455387,\n    \"max_depth\": 20,\n    \"max_leaves\": 12,\n    \"min_child_weight\": 16,\n    \"n_estimators\": 1667,\n    \"n_jobs\": -1,\n    \"random_state\": 42,\n    \"reg_alpha\": 39.352415706891264,\n    \"reg_lambda\": 75.44843704068275,\n    \"subsample\": 0.06566669853471274,\n    \"verbosity\": 0\n}\n\n# GANDALF Model Implementation\nfrom sklearn.base import BaseEstimator, RegressorMixin\nimport torch\nimport torch.nn as nn\nimport torch.optim as optim\nfrom torch.utils.data import DataLoader, TensorDataset\n\nclass GANDALF(BaseEstimator, RegressorMixin):\n    def __init__(self, n_estimators=100, learning_rate=0.01, max_depth=5, \n                 feature_fraction=0.8, bagging_fraction=0.8, lambda_reg=1.0,\n                 min_data_in_leaf=20, num_iterations=100, random_state=42):\n        self.n_estimators = n_estimators\n        self.learning_rate = learning_rate\n        self.max_depth = max_depth\n        self.feature_fraction = feature_fraction\n        self.bagging_fraction = bagging_fraction\n        self.lambda_reg = lambda_reg\n        self.min_data_in_leaf = min_data_in_leaf\n        self.num_iterations = num_iterations\n        self.random_state = random_state\n        self.device = torch.device('cuda' if torch.cuda.is_available() else 'cpu')\n        torch.manual_seed(random_state)\n        \n    def _build_network(self, input_dim):\n        \"\"\"Build the neural network architecture for GANDALF\"\"\"\n        class GANDALFNet(nn.Module):\n            def __init__(self, input_dim, hidden_dims=[256, 128, 64]):\n                super(GANDALFNet, self).__init__()\n                \n                layers = []\n                prev_dim = input_dim\n                \n                for hidden_dim in hidden_dims:\n                    layers.extend([\n                        nn.Linear(prev_dim, hidden_dim),\n                        nn.BatchNorm1d(hidden_dim),\n                        nn.ReLU(),\n                        nn.Dropout(0.3)\n                    ])\n                    prev_dim = hidden_dim\n                \n                layers.append(nn.Linear(prev_dim, 1))\n                \n                self.network = nn.Sequential(*layers)\n                \n            def forward(self, x):\n                return self.network(x)\n        \n        return GANDALFNet(input_dim).to(self.device)\n    \n    def fit(self, X, y):\n        # Convert to tensors\n        X_tensor = torch.FloatTensor(X.values if hasattr(X, 'values') else X).to(self.device)\n        y_tensor = torch.FloatTensor(y.values if hasattr(y, 'values') else y).reshape(-1, 1).to(self.device)\n        \n        # Build network\n        self.model = self._build_network(X_tensor.shape[1])\n        \n        # Create dataset and dataloader\n        dataset = TensorDataset(X_tensor, y_tensor)\n        dataloader = DataLoader(dataset, batch_size=1024, shuffle=True)\n        \n        # Optimizer\n        optimizer = optim.AdamW(self.model.parameters(), lr=self.learning_rate, weight_decay=self.lambda_reg)\n        criterion = nn.MSELoss()\n        \n        # Training loop\n        self.model.train()\n        for epoch in range(self.num_iterations):\n            total_loss = 0\n            for batch_X, batch_y in dataloader:\n                optimizer.zero_grad()\n                outputs = self.model(batch_X)\n                loss = criterion(outputs, batch_y)\n                loss.backward()\n                optimizer.step()\n                total_loss += loss.item()\n            \n            if (epoch + 1) % 20 == 0:\n                print(f'Epoch [{epoch+1}/{self.num_iterations}], Loss: {total_loss/len(dataloader):.4f}')\n        \n        return self\n    \n    def predict(self, X):\n        self.model.eval()\n        X_tensor = torch.FloatTensor(X.values if hasattr(X, 'values') else X).to(self.device)\n        \n        with torch.no_grad():\n            predictions = self.model(X_tensor).cpu().numpy().flatten()\n        \n        return predictions\n\n# AutoEncoder MLP\nfrom sklearn.base import BaseEstimator, RegressorMixin\nimport tensorflow as tf\nimport numpy as np\n\nclass AutoEncoderMLP(BaseEstimator, RegressorMixin):\n    def __init__(self, num_columns, hidden_units, dropout_rates, lr=1e-3, seed=42):\n        self.num_columns = num_columns\n        self.hidden_units = hidden_units\n        self.dropout_rates = dropout_rates\n        self.lr = lr\n        self.seed = seed\n        tf.random.set_seed(seed)\n        self.model = self._build_model()\n    \n    def _build_model(self):\n        inp = tf.keras.layers.Input(shape=(self.num_columns,))\n        x0 = tf.keras.layers.BatchNormalization()(inp)\n\n        encoder = tf.keras.layers.GaussianNoise(self.dropout_rates[0])(x0)\n        encoder = tf.keras.layers.Dense(self.hidden_units[0])(encoder)\n        encoder = tf.keras.layers.BatchNormalization()(encoder)\n        encoder = tf.keras.layers.Activation('swish')(encoder)\n\n        decoder = tf.keras.layers.Dropout(self.dropout_rates[1])(encoder)\n        decoder = tf.keras.layers.Dense(self.num_columns, name='decoder')(decoder)\n\n        x_reg = tf.keras.layers.Dense(self.hidden_units[1])(encoder)\n        x_reg = tf.keras.layers.BatchNormalization()(x_reg)\n        x_reg = tf.keras.layers.Activation('swish')(x_reg)\n        x_reg = tf.keras.layers.Dropout(self.dropout_rates[2])(x_reg)\n\n        out_reg = tf.keras.layers.Dense(1, activation='linear', name='target')(x_reg)\n\n        model = tf.keras.models.Model(inputs=inp, outputs=[decoder, out_reg])\n        model.compile(\n            optimizer=tf.keras.optimizers.Adam(learning_rate=self.lr),\n            loss={\"decoder\": tf.keras.losses.MeanSquaredError(),\n                  \"target\": tf.keras.losses.MeanSquaredError()},\n            loss_weights={\"decoder\": 0.3, \"target\": 1.0}\n        )\n        return model\n\n    def fit(self, X, y):\n        self.model.fit(\n            X, {\"decoder\": X, \"target\": y},\n            epochs=50,\n            batch_size=8192,\n            validation_split=0.2,\n            callbacks=[\n                tf.keras.callbacks.EarlyStopping(patience=10, restore_best_weights=True),\n                tf.keras.callbacks.ReduceLROnPlateau(patience=5)\n            ],\n            verbose=0\n        )\n        return self\n\n    def predict(self, X):\n        _, y_pred = self.model.predict(X, verbose=0)\n        return y_pred.flatten()\n\ndef run_pipeline_with_prophet_guided_compression(base_compression: float, \n                                               label_compression: float,\n                                               strategy_name: str,\n                                               n_stability_folds: int = 5):\n    \"\"\"\n    Run pipeline with Prophet-guided compression\n    \"\"\"\n    print(f\"\\n{'='*80}\")\n    print(f\"Running pipeline with Prophet-guided compression: {strategy_name}\")\n    print(f\"Base compression: {base_compression:.2f}, Label compression: {label_compression:.2f}\")\n    print(f\"{'='*80}\\n\")\n    \n    # Load data\n    train = pd.read_parquet(CFG.train_path).reset_index(drop=True)\n    test = pd.read_parquet(CFG.test_path).reset_index(drop=True)\n    \n    # Define features\n    X_FEATURES = ['X363', 'X321', 'X405', 'X730', 'X523', 'X756', 'X589', 'X462', 'X779',\n                    'X25', 'X532', 'X520', 'X329', 'X383', 'X751', 'X535', 'X639', 'X596', 'X761',\n                \"X752\", \"X287\", \"X298\", \"X759\", \"X302\", \"X55\", \"X56\", \"X52\", \"X303\", \"X51\",\n                \"X598\", \"X385\", \"X603\", \"X674\", \"X415\", \"X345\", \"X174\", \"X178\", \"X168\", \"X612\",\n                \"bid_qty\", \"ask_qty\", \"buy_qty\", \"sell_qty\"]\n    \n    selected_columns = X_FEATURES + [\"volume\"]\n    \n    # Select columns\n    train = train[selected_columns + [CFG.target]]\n    test = test[selected_columns]\n    \n    # Prepare data\n    X_train = train.drop(CFG.target, axis=1)\n    y_train = train[CFG.target]\n    X_test = test\n    \n    # Initialize and fit compressor\n    compressor = ProphetGuidedCompressor()\n    compressor.fit(X_train, y_train, base_compression=base_compression, \n                  label_compression=label_compression)\n    \n    # Apply compression\n    X_train_compressed, y_train_compressed = compressor.transform(\n        X_train, y_train, base_compression=base_compression\n    )\n    X_test_compressed, _ = compressor.transform(\n        X_test, None, base_compression=base_compression\n    )\n    \n    # Save compressor\n    compressor.save(f'compressor_{strategy_name}.pkl')\n    \n    # Visualize compression effects\n    fig, axes = plt.subplots(2, 2, figsize=(14, 10))\n    axes = axes.ravel()\n    \n    for idx, feat in enumerate(X_FEATURES[:4]):\n        ax = axes[idx]\n        \n        original = X_train[feat].values\n        compressed = X_train_compressed[feat].values\n        \n        # Remove NaN for plotting\n        mask = ~(np.isnan(original) | np.isnan(compressed))\n        \n        # Get compression profile\n        profile = compressor.feature_profiles.get(feat)\n        if profile:\n            # Color by compression strength\n            compression_strengths = profile.compression_map.predict(original[mask])\n            scatter = ax.scatter(original[mask], compressed[mask], \n                               c=compression_strengths, cmap='RdYlGn_r', \n                               alpha=0.5, s=10)\n            plt.colorbar(scatter, ax=ax, label='Compression')\n            \n            # Mark Prophet outliers if any\n            if profile.prophet_outlier_percentiles:\n                for p in profile.prophet_outlier_percentiles[:5]:  # Show first 5\n                    val = np.percentile(original[mask], p)\n                    ax.axvline(val, color='red', linestyle=':', alpha=0.3)\n        else:\n            ax.scatter(original[mask], compressed[mask], alpha=0.5, s=10)\n        \n        # Reference line\n        ax.plot([original[mask].min(), original[mask].max()], \n               [original[mask].min(), original[mask].max()], 'b--', alpha=0.5)\n        \n        ax.set_title(f'{feat}')\n        ax.set_xlabel('Original')\n        ax.set_ylabel('Compressed')\n    \n    plt.tight_layout()\n    plt.savefig(f'prophet_guided_compression_{strategy_name}.png', dpi=150, bbox_inches='tight')\n    plt.close()\n    \n    # Reduce memory\n    X_train_compressed = reduce_mem_usage(X_train_compressed, \"train\")\n    X_test_compressed = reduce_mem_usage(X_test_compressed, \"test\")\n    \n    # Run stability analysis with different fold configurations\n    stability_results = {}\n    all_test_predictions = []\n    \n    for fold_config in range(n_stability_folds):\n        print(f\"\\nStability fold configuration {fold_config + 1}/{n_stability_folds}\")\n        \n        # Create different fold splits\n        if fold_config == 0:\n            cv = KFold(n_splits=CFG.n_folds, shuffle=True, random_state=CFG.seed)\n        elif fold_config == 1:\n            cv = KFold(n_splits=CFG.n_folds, shuffle=True, random_state=CFG.seed + 1)\n        elif fold_config == 2:\n            cv = KFold(n_splits=8, shuffle=True, random_state=CFG.seed + 2)\n        elif fold_config == 3:\n            cv = TimeSeriesSplit(n_splits=min(5, CFG.n_folds))\n        else:\n            cv = KFold(n_splits=CFG.n_folds, shuffle=True, random_state=CFG.seed + fold_config)\n        \n        # Initialize storage for this fold configuration\n        scores = {}\n        test_preds = {}\n        \n        # Train models\n        # LightGBM (gbdt)\n        lgbm_trainer = Trainer(\n            LGBMRegressor(**lgbm_params),\n            cv=cv,\n            metric=_pearsonr,\n            task=\"regression\",\n            metric_precision=6\n        )\n        lgbm_trainer.fit(X_train_compressed, y_train_compressed)\n        scores[\"LightGBM (gbdt)\"] = lgbm_trainer.fold_scores\n        test_preds[\"LightGBM (gbdt)\"] = lgbm_trainer.predict(X_test_compressed)\n        \n        # LightGBM (goss)\n        lgbm_goss_trainer = Trainer(\n            LGBMRegressor(**lgbm_goss_params),\n            cv=cv,\n            metric=_pearsonr,\n            task=\"regression\",\n            metric_precision=6\n        )\n        lgbm_goss_trainer.fit(X_train_compressed, y_train_compressed)\n        scores[\"LightGBM (goss)\"] = lgbm_goss_trainer.fold_scores\n        test_preds[\"LightGBM (goss)\"] = lgbm_goss_trainer.predict(X_test_compressed)\n        \n        # XGBoost\n        xgb_trainer = Trainer(\n            XGBRegressor(**xgb_params),\n            cv=cv,\n            metric=_pearsonr,\n            task=\"regression\",\n            metric_precision=6\n        )\n        xgb_trainer.fit(X_train_compressed, y_train_compressed)\n        scores[\"XGBoost\"] = xgb_trainer.fold_scores\n        test_preds[\"XGBoost\"] = xgb_trainer.predict(X_test_compressed)\n        \n        # For full pipeline, add GANDALF and AutoEncoder\n        if fold_config == 0:  # Only run once to save time\n            # GANDALF\n            print(f\"\\nTraining GANDALF for {strategy_name}...\")\n            gandalf_model = GANDALF(\n                n_estimators=100,\n                learning_rate=0.001,\n                max_depth=5,\n                feature_fraction=0.8,\n                bagging_fraction=0.8,\n                lambda_reg=0.1,\n                min_data_in_leaf=20,\n                num_iterations=50,\n                random_state=42\n            )\n            gandalf_trainer = Trainer(\n                gandalf_model,\n                cv=cv,\n                metric=_pearsonr,\n                task=\"regression\",\n                metric_precision=6\n            )\n            gandalf_trainer.fit(X_train_compressed, y_train_compressed)\n            scores[\"GANDALF\"] = gandalf_trainer.fold_scores\n            test_preds[\"GANDALF\"] = gandalf_trainer.predict(X_test_compressed)\n            \n            # Create ensemble predictions\n            X_ensemble = pd.DataFrame({\n                \"LightGBM (gbdt)\": lgbm_trainer.oof_preds,\n                \"LightGBM (goss)\": lgbm_goss_trainer.oof_preds,\n                \"XGBoost\": xgb_trainer.oof_preds,\n                \"GANDALF\": gandalf_trainer.oof_preds\n            })\n            X_test_ensemble = pd.DataFrame(test_preds)\n            \n            # AutoEncoder\n            print(f\"\\nTraining AutoEncoder for {strategy_name}...\")\n            ae_model = AutoEncoderMLP(\n                num_columns=X_ensemble.shape[1],\n                hidden_units=[128, 128],\n                dropout_rates=[0.05, 0.1, 0.2],\n                lr=1e-3,\n                seed=42\n            )\n            ae_trainer = Trainer(\n                ae_model,\n                cv=cv,\n                metric=_pearsonr,\n                task=\"regression\",\n                metric_precision=6\n            )\n            ae_trainer.fit(X_ensemble, y_train_compressed)\n            scores[\"AutoEncoder\"] = ae_trainer.fold_scores\n            test_preds[\"AutoEncoder\"] = ae_trainer.predict(X_test_ensemble)\n        \n        # Calculate weighted predictions for this fold\n        if \"GANDALF\" in test_preds and \"AutoEncoder\" in test_preds:\n            ensemble_weights = {\n                \"LightGBM (gbdt)\": 0.285,\n                \"LightGBM (goss)\": 0.285,\n                \"XGBoost\": 0.285,\n                \"GANDALF\": 0.05,\n                \"AutoEncoder\": 0.095\n            }\n        else:\n            ensemble_weights = {\n                \"LightGBM (gbdt)\": 0.4,\n                \"LightGBM (goss)\": 0.3,\n                \"XGBoost\": 0.3\n            }\n        \n        weighted_test_pred = np.zeros(len(X_test))\n        for model_name, weight in ensemble_weights.items():\n            if model_name in test_preds:\n                weighted_test_pred += weight * test_preds[model_name]\n        \n        all_test_predictions.append(weighted_test_pred)\n        stability_results[f'fold_{fold_config}'] = scores\n    \n    # Calculate overall stability metrics\n    all_scores_flat = {}\n    for fold_results in stability_results.values():\n        for model, scores in fold_results.items():\n            if model not in all_scores_flat:\n                all_scores_flat[model] = []\n            all_scores_flat[model].extend(scores)\n    \n    stability_metrics = calculate_stability_metrics(all_scores_flat)\n    \n    # Average predictions across stability folds\n    final_predictions = np.mean(all_test_predictions, axis=0)\n    \n    # Clean up memory\n    del X_train, y_train, X_test, train, test\n    gc.collect()\n    \n    return final_predictions, stability_metrics, compressor, all_test_predictions\n\n# Main execution\nif __name__ == \"__main__\":\n    # Compression grid\n    compression_grid = [\n        # (base_compression, label_compression, name)\n        (0.0, 0.0, \"baseline_no_compression\"),\n        (0.2, 0.0, \"mild_20pct\"),\n        (0.3, 0.0, \"moderate_30pct\"),\n        (0.4, 0.0, \"strong_40pct\"),\n        (0.3, 0.05, \"balanced_30_5pct\"),\n        (0.35, 0.05, \"balanced_35_5pct\"),\n        (0.25, 0.0, \"prophet_guided_25pct\"),\n        (0.35, 0.0, \"prophet_guided_35pct\"),\n    ]\n    \n    # Store results\n    all_results = {}\n    \n    # Run grid search with stability analysis\n    for base_comp, label_comp, name in compression_grid:\n        try:\n            predictions, stability_metrics, compressor, fold_predictions = \\\n                run_pipeline_with_prophet_guided_compression(\n                    base_comp, label_comp, name, \n                    n_stability_folds=CFG.n_stability_folds\n                )\n            \n            all_results[name] = {\n                'predictions': predictions,\n                'stability_metrics': stability_metrics,\n                'compressor': compressor,\n                'fold_predictions': fold_predictions\n            }\n            \n            # Save individual submission\n            sub = pd.read_csv(CFG.sample_sub_path)\n            sub[\"prediction\"] = predictions\n            sub.to_csv(f\"submission_{name}.csv\", index=False)\n            print(f\"\\nSaved submission_{name}.csv\")\n            print(f\"Stability score: {stability_metrics.stability_score:.4f}\")\n            print(f\"Mean CV: {stability_metrics.mean_cv_score:.4f} ± {stability_metrics.std_cv_score:.4f}\")\n            \n        except Exception as e:\n            print(f\"\\nError in strategy {name}: {str(e)}\")\n            import traceback\n            traceback.print_exc()\n            continue\n    \n    # Select top 3 most stable strategies\n    print(\"\\n\" + \"=\"*80)\n    print(\"SELECTING TOP 3 MOST STABLE STRATEGIES\")\n    print(\"=\"*80)\n    \n    stability_scores = {\n        name: results['stability_metrics'].stability_score \n        for name, results in all_results.items()\n    }\n    \n    top_3_strategies = sorted(stability_scores.items(), key=lambda x: x[1], reverse=True)[:3]\n    \n    print(\"\\nTop 3 most stable strategies:\")\n    for i, (strategy, score) in enumerate(top_3_strategies, 1):\n        metrics = all_results[strategy]['stability_metrics']\n        print(f\"{i}. {strategy}: stability={score:.4f}, \"\n              f\"mean={metrics.mean_cv_score:.4f}, std={metrics.std_cv_score:.4f}\")\n    \n    # Create final ensemble from top 3 strategies\n    print(\"\\nCreating final ensemble from top 3 strategies...\")\n    \n    # Calculate dynamic weights\n    top_predictions = {}\n    top_stability_scores = {}\n    \n    for strategy, score in top_3_strategies:\n        top_predictions[strategy] = all_results[strategy]['predictions']\n        top_stability_scores[strategy] = score\n    \n    # Weight by stability score\n    total_score = sum(top_stability_scores.values())\n    optimal_weights = {s: score/total_score for s, score in top_stability_scores.items()}\n    \n    print(\"\\nOptimal ensemble weights:\")\n    for strategy, weight in optimal_weights.items():\n        print(f\"  {strategy}: {weight:.3f}\")\n    \n    # Create final ensemble\n    final_ensemble = np.zeros_like(list(top_predictions.values())[0])\n    for strategy, weight in optimal_weights.items():\n        final_ensemble += weight * top_predictions[strategy]\n    \n    # Save final ensemble\n    sub = pd.read_csv(CFG.sample_sub_path)\n    sub[\"prediction\"] = final_ensemble\n    sub.to_csv(\"submission_prophet_guided_ensemble.csv\", index=False)\n    print(\"\\nSaved submission_prophet_guided_ensemble.csv\")\n    \n    # Create visualizations\n    # 1. Stability comparison plot\n    plt.figure(figsize=(12, 8))\n    strategies = list(stability_scores.keys())\n    scores = list(stability_scores.values())\n    means = [all_results[s]['stability_metrics'].mean_cv_score for s in strategies]\n    stds = [all_results[s]['stability_metrics'].std_cv_score for s in strategies]\n    \n    x = np.arange(len(strategies))\n    bars = plt.bar(x, scores, alpha=0.7, label='Stability Score', color='steelblue')\n    \n    # Highlight top 3\n    top_indices = [strategies.index(s) for s, _ in top_3_strategies]\n    for idx in top_indices:\n        bars[idx].set_color('gold')\n        bars[idx].set_edgecolor('black')\n        bars[idx].set_linewidth(2)\n    \n    plt.errorbar(x, means, yerr=stds, fmt='o', color='darkred', \n                capsize=5, capthick=2, label='Mean ± Std')\n    \n    plt.xticks(x, strategies, rotation=45, ha='right')\n    plt.ylabel('Score')\n    plt.title('Prophet-Guided Compression Strategy Stability Analysis')\n    plt.legend()\n    plt.grid(True, alpha=0.3)\n    plt.tight_layout()\n    plt.savefig('prophet_guided_stability_analysis.png', dpi=300, bbox_inches='tight')\n    plt.show()\n    \n    # 2. Prophet influence visualization\n    fig, axes = plt.subplots(2, 2, figsize=(14, 10))\n    axes = axes.ravel()\n    \n    # Select best strategy\n    best_strategy = top_3_strategies[0][0]\n    compressor = all_results[best_strategy]['compressor']\n    \n    features_to_analyze = ['X363', 'bid_qty', 'X523', 'volume']\n    \n    for idx, feat in enumerate(features_to_analyze[:4]):\n        ax = axes[idx]\n        \n        if feat in compressor.feature_profiles:\n            profile = compressor.feature_profiles[feat]\n            \n            # Get feature values\n            train = pd.read_parquet(CFG.train_path)\n            if feat in train.columns:\n                values = train[feat].values[:10000]  # Sample for visualization\n                values = values[~np.isnan(values)]\n                \n                # Get compression strengths\n                compressions = profile.compression_map.predict(values)\n                \n                # Create 2D histogram\n                h = ax.hist2d(values, compressions, bins=50, cmap='YlOrRd')\n                plt.colorbar(h[3], ax=ax)\n                \n                # Mark percentiles\n                for p in [10, 50, 90]:\n                    ax.axvline(profile.percentiles[p], color='blue', linestyle=':', alpha=0.5)\n                \n                # Mark Prophet outliers\n                if profile.prophet_outlier_percentiles:\n                    for p in profile.prophet_outlier_percentiles[:3]:\n                        val = np.percentile(values, p)\n                        ax.axvline(val, color='red', linestyle='--', alpha=0.7, \n                                  label='Prophet outlier' if idx == 0 else '')\n                \n                ax.set_xlabel(f'{feat} Value')\n                ax.set_ylabel('Compression Strength')\n                ax.set_title(f'Prophet-Guided Compression: {feat}')\n                if idx == 0:\n                    ax.legend()\n    \n    plt.tight_layout()\n    plt.savefig('prophet_compression_mapping.png', dpi=300, bbox_inches='tight')\n    plt.show()\n    \n    # Save comprehensive summary\n    summary_data = []\n    for name, results in all_results.items():\n        metrics = results['stability_metrics']\n        summary_data.append({\n            'Strategy': name,\n            'Mean_CV': metrics.mean_cv_score,\n            'Std_CV': metrics.std_cv_score,\n            'Stability_Score': metrics.stability_score,\n            'Inter_Fold_Correlation': metrics.inter_fold_correlation,\n            'Is_Top_3': name in [s for s, _ in top_3_strategies],\n            'Final_Weight': optimal_weights.get(name, 0)\n        })\n    \n    summary_df = pd.DataFrame(summary_data)\n    summary_df = summary_df.sort_values('Stability_Score', ascending=False)\n    summary_df.to_csv('prophet_guided_compression_summary.csv', index=False)\n    \n    print(\"\\n\" + \"=\"*80)\n    print(\"PROPHET-GUIDED COMPRESSION PIPELINE COMPLETED!\")\n    print(\"=\"*80)\n    print(\"Key advantages:\")\n    print(\"- Prophet identifies temporal outliers during training\")\n    print(\"- Learned compression mappings work without timestamps on test data\")\n    print(\"- Combines temporal insights with distribution-based compression\")\n    print(\"- Smooth isotonic regression ensures monotonic compression\")\n    print(\"\\nGenerated files:\")\n    print(\"- Individual submissions for each strategy\")\n    print(\"- submission_prophet_guided_ensemble.csv (TOP 3 ENSEMBLE)\")\n    print(\"- Prophet-guided compression profiles (.pkl files)\")\n    print(\"- prophet_guided_stability_analysis.png\")\n    print(\"- prophet_compression_mapping.png\")\n    print(\"- prophet_guided_compression_summary.csv\")\n    print(\"=\"*80)","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}