{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":96164,"databundleVersionId":12993472,"sourceType":"competition"}],"dockerImageVersionId":31089,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-07-25T01:44:23.871831Z","iopub.execute_input":"2025-07-25T01:44:23.872074Z","iopub.status.idle":"2025-07-25T01:44:26.264065Z","shell.execute_reply.started":"2025-07-25T01:44:23.872052Z","shell.execute_reply":"2025-07-25T01:44:26.260430Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Advanced Bidirectional Intelligent Compression with Quartile Strategies\n# This version can expand or compress different regions of the distribution differently\n\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom scipy.stats import pearsonr, kurtosis, skew\nfrom scipy.interpolate import UnivariateSpline, PchipInterpolator\nfrom scipy.optimize import minimize\nfrom sklearn.isotonic import IsotonicRegression\nfrom sklearn.ensemble import RandomForestRegressor\nfrom lightgbm import LGBMRegressor\nfrom sklearn.model_selection import KFold\nimport warnings\nwarnings.filterwarnings('ignore')\n\nplt.style.use('seaborn-v0_8-darkgrid')\nsns.set_palette(\"husl\")\n\nprint(\"=\"*80)\nprint(\"ADVANCED BIDIRECTIONAL COMPRESSION WITH QUARTILE STRATEGIES\")\nprint(\"=\"*80)\nprint(\"\\nThis enhanced system features:\")\nprint(\"1. Different compression algorithms per quartile\")\nprint(\"2. Adaptive gradient curves that can expand or compress\")\nprint(\"3. Train-CV gap minimization for noise verification\")\nprint(\"4. Flexible n-tile segmentation (not just quartiles)\")\nprint(\"=\"*80)\n\nclass BidirectionalQuartileCompressor:\n    \"\"\"\n    Advanced compressor that can apply different transformations to different\n    regions of the distribution, including expansion of compressed regions\n    \"\"\"\n    \n    def __init__(self, n_segments=4, optimization_metric='cv_gap'):\n        self.n_segments = n_segments\n        self.optimization_metric = optimization_metric\n        self.segment_profiles = {}\n        self.transformation_curves = {}\n        self.feature_cv_gaps = {}\n        \n    def segment_data(self, data):\n        \"\"\"Segment data into n quantile-based segments\"\"\"\n        # Remove NaN values\n        clean_data = data[~np.isnan(data)]\n        \n        # Calculate segment boundaries\n        percentiles = np.linspace(0, 100, self.n_segments + 1)\n        boundaries = np.percentile(clean_data, percentiles)\n        \n        # Ensure unique boundaries\n        boundaries = np.unique(boundaries)\n        if len(boundaries) < 2:\n            boundaries = np.array([clean_data.min(), clean_data.max()])\n        \n        return boundaries\n    \n    def create_gradient_curve(self, x_points, y_points, method='spline'):\n        \"\"\"Create smooth transformation curve from control points\"\"\"\n        if method == 'spline':\n            # Cubic spline for smooth curves\n            spline = UnivariateSpline(x_points, y_points, s=0.1, k=3)\n            return spline\n        elif method == 'pchip':\n            # Monotonic cubic interpolation\n            pchip = PchipInterpolator(x_points, y_points)\n            return pchip\n        elif method == 'isotonic':\n            # Isotonic regression for monotonic transformation\n            iso = IsotonicRegression()\n            iso.fit(x_points, y_points)\n            return iso\n        else:\n            # Linear interpolation as fallback\n            return lambda x: np.interp(x, x_points, y_points)\n    \n    def generate_random_gradient(self, segment_range, expansion_allowed=True):\n        \"\"\"Generate random gradient parameters for a segment\"\"\"\n        min_val, max_val = segment_range\n        \n        # Random transformation parameters\n        if expansion_allowed:\n            # Allow both compression and expansion\n            scale_factor = np.random.uniform(0.5, 2.0)  # 0.5 = compress by half, 2.0 = expand by 2x\n            shift_factor = np.random.uniform(-0.2, 0.2) * (max_val - min_val)\n        else:\n            # Only compression\n            scale_factor = np.random.uniform(0.5, 1.0)\n            shift_factor = 0\n        \n        # Random curve shape\n        curve_type = np.random.choice(['linear', 'sigmoid', 'power', 'log'])\n        curve_strength = np.random.uniform(0.1, 2.0)\n        \n        return {\n            'scale': scale_factor,\n            'shift': shift_factor,\n            'curve_type': curve_type,\n            'strength': curve_strength\n        }\n    \n    def apply_segment_transformation(self, data, segment_params, segment_range):\n        \"\"\"Apply transformation to data within a segment\"\"\"\n        min_val, max_val = segment_range\n        mask = (data >= min_val) & (data <= max_val)\n        \n        if not np.any(mask):\n            return data\n        \n        segment_data = data[mask]\n        \n        # Normalize to [0, 1] within segment\n        if max_val > min_val:\n            normalized = (segment_data - min_val) / (max_val - min_val)\n        else:\n            normalized = np.zeros_like(segment_data)\n        \n        # Apply curve transformation\n        curve_type = segment_params['curve_type']\n        strength = segment_params['strength']\n        \n        if curve_type == 'linear':\n            transformed = normalized\n        elif curve_type == 'sigmoid':\n            # Sigmoid transformation\n            transformed = 1 / (1 + np.exp(-strength * (normalized - 0.5) * 4))\n        elif curve_type == 'power':\n            # Power transformation\n            transformed = np.power(normalized, strength)\n        elif curve_type == 'log':\n            # Log transformation\n            transformed = np.log1p(normalized * strength) / np.log1p(strength)\n        \n        # Apply scale and shift\n        transformed = transformed * segment_params['scale'] + segment_params['shift']\n        \n        # Denormalize back to original scale\n        result = data.copy()\n        result[mask] = transformed * (max_val - min_val) + min_val\n        \n        return result\n    \n    def evaluate_transformation(self, X, y, feature_idx, transformation_params):\n        \"\"\"Evaluate transformation quality using train-CV gap\"\"\"\n        # Apply transformation\n        X_transformed = X.copy()\n        feature_data = X.iloc[:, feature_idx].values\n        \n        # Apply transformations for each segment\n        for seg_idx, (seg_range, seg_params) in enumerate(transformation_params):\n            feature_data = self.apply_segment_transformation(\n                feature_data, seg_params, seg_range\n            )\n        \n        X_transformed.iloc[:, feature_idx] = feature_data\n        \n        # Evaluate using simple model\n        model = LGBMRegressor(n_estimators=50, random_state=42, verbose=-1)\n        \n        # Calculate train score\n        model.fit(X_transformed, y)\n        train_pred = model.predict(X_transformed)\n        train_score = pearsonr(y, train_pred)[0]\n        \n        # Calculate CV score\n        cv_scores = []\n        kf = KFold(n_splits=3, shuffle=True, random_state=42)\n        \n        for train_idx, val_idx in kf.split(X_transformed):\n            model_cv = LGBMRegressor(n_estimators=50, random_state=42, verbose=-1)\n            model_cv.fit(X_transformed.iloc[train_idx], y.iloc[train_idx])\n            val_pred = model_cv.predict(X_transformed.iloc[val_idx])\n            cv_scores.append(pearsonr(y.iloc[val_idx], val_pred)[0])\n        \n        cv_score = np.mean(cv_scores)\n        \n        # Calculate gap (lower is better)\n        gap = train_score - cv_score\n        \n        return {\n            'train_score': train_score,\n            'cv_score': cv_score,\n            'gap': gap,\n            'gap_reduction': -gap  # Negative because we minimize\n        }\n    \n    def optimize_feature_transformation(self, X, y, feature_idx, feature_name, \n                                      n_iterations=20, visualize=True):\n        \"\"\"Optimize transformation for a single feature\"\"\"\n        print(f\"\\n{'='*60}\")\n        print(f\"🔧 OPTIMIZING TRANSFORMATION FOR: {feature_name}\")\n        print(f\"{'='*60}\")\n        \n        feature_data = X.iloc[:, feature_idx].values\n        \n        # Segment the data\n        boundaries = self.segment_data(feature_data)\n        segments = [(boundaries[i], boundaries[i+1]) for i in range(len(boundaries)-1)]\n        \n        print(f\"\\n📊 Data segmented into {len(segments)} regions:\")\n        for i, (min_val, max_val) in enumerate(segments):\n            count = np.sum((feature_data >= min_val) & (feature_data <= max_val))\n            print(f\"   Segment {i+1}: [{min_val:.3f}, {max_val:.3f}] - {count} samples\")\n        \n        # Initialize with baseline (no transformation)\n        baseline_params = [(seg, {'scale': 1.0, 'shift': 0, 'curve_type': 'linear', 'strength': 1}) \n                          for seg in segments]\n        baseline_eval = self.evaluate_transformation(X, y, feature_idx, baseline_params)\n        \n        print(f\"\\n📈 Baseline performance:\")\n        print(f\"   Train score: {baseline_eval['train_score']:.4f}\")\n        print(f\"   CV score: {baseline_eval['cv_score']:.4f}\")\n        print(f\"   Train-CV gap: {baseline_eval['gap']:.4f}\")\n        \n        # Optimization loop\n        best_params = baseline_params\n        best_eval = baseline_eval\n        history = [baseline_eval]\n        \n        print(f\"\\n🔄 Running {n_iterations} optimization iterations...\")\n        \n        for iteration in range(n_iterations):\n            # Generate random variations\n            candidates = []\n            \n            for _ in range(10):  # Test 10 random variations per iteration\n                # Randomly modify segments\n                new_params = []\n                for seg_range, seg_params in best_params:\n                    if np.random.random() < 0.7:  # 70% chance to modify segment\n                        new_seg_params = self.generate_random_gradient(seg_range)\n                    else:\n                        new_seg_params = seg_params.copy()\n                    new_params.append((seg_range, new_seg_params))\n                \n                # Evaluate\n                eval_result = self.evaluate_transformation(X, y, feature_idx, new_params)\n                candidates.append((new_params, eval_result))\n            \n            # Select best candidate\n            if self.optimization_metric == 'cv_gap':\n                best_candidate = min(candidates, key=lambda x: x[1]['gap'])\n            else:  # cv_score\n                best_candidate = max(candidates, key=lambda x: x[1]['cv_score'])\n            \n            # Update if improved\n            if (self.optimization_metric == 'cv_gap' and best_candidate[1]['gap'] < best_eval['gap']) or \\\n               (self.optimization_metric == 'cv_score' and best_candidate[1]['cv_score'] > best_eval['cv_score']):\n                best_params = best_candidate[0]\n                best_eval = best_candidate[1]\n                history.append(best_eval)\n                \n                if (iteration + 1) % 5 == 0:\n                    print(f\"   Iteration {iteration+1}: Gap={best_eval['gap']:.4f}, \"\n                          f\"CV={best_eval['cv_score']:.4f}\")\n        \n        print(f\"\\n✅ Optimization complete!\")\n        print(f\"   Final train score: {best_eval['train_score']:.4f}\")\n        print(f\"   Final CV score: {best_eval['cv_score']:.4f}\")\n        print(f\"   Final gap: {best_eval['gap']:.4f}\")\n        print(f\"   Gap reduction: {baseline_eval['gap'] - best_eval['gap']:.4f}\")\n        \n        # Store results\n        self.segment_profiles[feature_name] = {\n            'segments': segments,\n            'parameters': best_params,\n            'baseline_eval': baseline_eval,\n            'optimized_eval': best_eval,\n            'history': history\n        }\n        \n        # Visualize if requested\n        if visualize:\n            self._visualize_transformation(feature_data, feature_name, best_params)\n        \n        return best_params\n    \n    def _visualize_transformation(self, original_data, feature_name, transformation_params):\n        \"\"\"Visualize the transformation curve and effects\"\"\"\n        fig, axes = plt.subplots(2, 2, figsize=(15, 10))\n        fig.suptitle(f'Bidirectional Transformation for {feature_name}', fontsize=16)\n        \n        # Remove NaN\n        clean_data = original_data[~np.isnan(original_data)]\n        \n        # 1. Transformation curve\n        ax = axes[0, 0]\n        \n        # Create smooth curve for visualization\n        x_range = np.linspace(clean_data.min(), clean_data.max(), 1000)\n        y_transformed = x_range.copy()\n        \n        for seg_range, seg_params in transformation_params:\n            y_transformed = self.apply_segment_transformation(\n                y_transformed, seg_params, seg_range\n            )\n        \n        ax.plot(x_range, y_transformed, 'b-', linewidth=2, label='Transformation')\n        ax.plot(x_range, x_range, 'r--', alpha=0.5, label='Identity (no change)')\n        \n        # Mark segment boundaries\n        for seg_range, _ in transformation_params:\n            ax.axvline(seg_range[0], color='gray', linestyle=':', alpha=0.5)\n        \n        ax.set_xlabel('Original Value')\n        ax.set_ylabel('Transformed Value')\n        ax.set_title('Transformation Function')\n        ax.legend()\n        ax.grid(True, alpha=0.3)\n        \n        # 2. Segment details\n        ax = axes[0, 1]\n        segment_info = []\n        \n        for i, (seg_range, seg_params) in enumerate(transformation_params):\n            segment_info.append({\n                'Segment': f'S{i+1}',\n                'Range': f'[{seg_range[0]:.2f}, {seg_range[1]:.2f}]',\n                'Scale': seg_params['scale'],\n                'Type': seg_params['curve_type'],\n                'Action': 'Expand' if seg_params['scale'] > 1 else 'Compress'\n            })\n        \n        segment_df = pd.DataFrame(segment_info)\n        \n        # Create text visualization\n        text = \"Segment Transformations:\\n\\n\"\n        for _, row in segment_df.iterrows():\n            text += f\"{row['Segment']}: {row['Range']}\\n\"\n            text += f\"  • Action: {row['Action']} ({row['Scale']:.2f}x)\\n\"\n            text += f\"  • Curve: {row['Type']}\\n\\n\"\n        \n        ax.text(0.1, 0.9, text, transform=ax.transAxes, fontsize=10,\n                verticalalignment='top', fontfamily='monospace',\n                bbox=dict(boxstyle='round', facecolor='wheat', alpha=0.5))\n        ax.axis('off')\n        \n        # 3. Before/After distribution\n        ax = axes[1, 0]\n        \n        # Apply transformation\n        transformed_data = clean_data.copy()\n        for seg_range, seg_params in transformation_params:\n            transformed_data = self.apply_segment_transformation(\n                transformed_data, seg_params, seg_range\n            )\n        \n        ax.hist(clean_data, bins=50, alpha=0.5, label='Original', density=True, color='blue')\n        ax.hist(transformed_data, bins=50, alpha=0.5, label='Transformed', density=True, color='red')\n        \n        # Mark quartiles\n        for p in [25, 50, 75]:\n            orig_val = np.percentile(clean_data, p)\n            trans_val = np.percentile(transformed_data, p)\n            ax.axvline(orig_val, color='blue', linestyle=':', alpha=0.5)\n            ax.axvline(trans_val, color='red', linestyle=':', alpha=0.5)\n        \n        ax.set_xlabel('Value')\n        ax.set_ylabel('Density')\n        ax.set_title('Distribution Comparison')\n        ax.legend()\n        \n        # 4. Optimization history\n        ax = axes[1, 1]\n        \n        if feature_name in self.segment_profiles:\n            history = self.segment_profiles[feature_name]['history']\n            \n            iterations = range(len(history))\n            gaps = [h['gap'] for h in history]\n            cv_scores = [h['cv_score'] for h in history]\n            \n            ax2 = ax.twinx()\n            \n            line1 = ax.plot(iterations, gaps, 'b-', marker='o', label='Train-CV Gap')\n            line2 = ax2.plot(iterations, cv_scores, 'r-', marker='s', label='CV Score')\n            \n            ax.set_xlabel('Iteration')\n            ax.set_ylabel('Train-CV Gap', color='b')\n            ax2.set_ylabel('CV Score', color='r')\n            ax.set_title('Optimization Progress')\n            \n            # Combine legends\n            lines = line1 + line2\n            labels = [l.get_label() for l in lines]\n            ax.legend(lines, labels, loc='center right')\n        \n        plt.tight_layout()\n        plt.savefig(f'bidirectional_transformation_{feature_name}.png', dpi=150, bbox_inches='tight')\n        plt.show()\n    \n    def fit(self, X, y, feature_subset=None, n_iterations=20):\n        \"\"\"Fit the compressor by optimizing each feature\"\"\"\n        print(\"\\n\" + \"=\"*80)\n        print(\"🚀 FITTING BIDIRECTIONAL QUARTILE COMPRESSOR\")\n        print(\"=\"*80)\n        \n        if feature_subset is None:\n            feature_subset = list(range(X.shape[1]))\n        \n        for idx in feature_subset[:5]:  # Limit to first 5 for demonstration\n            feature_name = X.columns[idx] if hasattr(X, 'columns') else f'Feature_{idx}'\n            self.optimize_feature_transformation(X, y, idx, feature_name, n_iterations)\n        \n        return self\n    \n    def transform(self, X):\n        \"\"\"Apply learned transformations\"\"\"\n        X_transformed = X.copy()\n        \n        for feature_name, profile in self.segment_profiles.items():\n            if hasattr(X, 'columns') and feature_name in X.columns:\n                idx = list(X.columns).index(feature_name)\n            else:\n                continue\n            \n            feature_data = X_transformed.iloc[:, idx].values\n            \n            for seg_range, seg_params in profile['parameters']:\n                feature_data = self.apply_segment_transformation(\n                    feature_data, seg_params, seg_range\n                )\n            \n            X_transformed.iloc[:, idx] = feature_data\n        \n        return X_transformed\n\n# Advanced evaluation with multiple strategies\ndef compare_compression_strategies_advanced(X_train, y_train):\n    \"\"\"Compare different advanced compression strategies\"\"\"\n    print(\"\\n\" + \"=\"*80)\n    print(\"🔬 COMPARING ADVANCED COMPRESSION STRATEGIES\")\n    print(\"=\"*80)\n    \n    strategies = {\n        'Baseline (No Compression)': {\n            'type': 'none'\n        },\n        'Traditional Uniform': {\n            'type': 'uniform',\n            'strength': 0.3\n        },\n        'Quartile-Based (4 segments)': {\n            'type': 'bidirectional',\n            'n_segments': 4,\n            'metric': 'cv_gap'\n        },\n        'Decile-Based (10 segments)': {\n            'type': 'bidirectional', \n            'n_segments': 10,\n            'metric': 'cv_gap'\n        },\n        'CV-Score Optimized': {\n            'type': 'bidirectional',\n            'n_segments': 5,\n            'metric': 'cv_score'\n        }\n    }\n    \n    results = {}\n    \n    for name, config in strategies.items():\n        print(f\"\\n\\n{'='*60}\")\n        print(f\"Testing strategy: {name}\")\n        print(f\"{'='*60}\")\n        \n        if config['type'] == 'none':\n            X_transformed = X_train\n            \n        elif config['type'] == 'uniform':\n            # Simple uniform compression\n            from sklearn.preprocessing import RobustScaler\n            scaler = RobustScaler()\n            X_scaled = pd.DataFrame(\n                scaler.fit_transform(X_train),\n                columns=X_train.columns\n            )\n            X_transformed = X_scaled * (1 - config['strength']) + X_train * config['strength']\n            \n        elif config['type'] == 'bidirectional':\n            compressor = BidirectionalQuartileCompressor(\n                n_segments=config['n_segments'],\n                optimization_metric=config['metric']\n            )\n            # Fit on subset of features for speed\n            compressor.fit(X_train, y_train, feature_subset=list(range(10)), n_iterations=10)\n            X_transformed = compressor.transform(X_train)\n        \n        # Evaluate\n        model = LGBMRegressor(n_estimators=100, random_state=42, verbose=-1)\n        \n        # Training score\n        model.fit(X_transformed, y_train)\n        train_pred = model.predict(X_transformed)\n        train_score = pearsonr(y_train, train_pred)[0]\n        \n        # CV score\n        cv_scores = []\n        kf = KFold(n_splits=5, shuffle=True, random_state=42)\n        \n        for train_idx, val_idx in kf.split(X_transformed):\n            model_cv = LGBMRegressor(n_estimators=100, random_state=42, verbose=-1)\n            model_cv.fit(X_transformed.iloc[train_idx], y_train.iloc[train_idx])\n            val_pred = model_cv.predict(X_transformed.iloc[val_idx])\n            cv_scores.append(pearsonr(y_train.iloc[val_idx], val_pred)[0])\n        \n        cv_score = np.mean(cv_scores)\n        cv_std = np.std(cv_scores)\n        gap = train_score - cv_score\n        \n        results[name] = {\n            'train_score': train_score,\n            'cv_score': cv_score,\n            'cv_std': cv_std,\n            'gap': gap\n        }\n        \n        print(f\"\\n📊 Results:\")\n        print(f\"   Train Score: {train_score:.4f}\")\n        print(f\"   CV Score: {cv_score:.4f} ± {cv_std:.4f}\")\n        print(f\"   Train-CV Gap: {gap:.4f}\")\n    \n    # Visualization\n    fig, axes = plt.subplots(1, 2, figsize=(15, 6))\n    \n    # 1. Scores comparison\n    ax = axes[0]\n    strategies_list = list(results.keys())\n    train_scores = [results[s]['train_score'] for s in strategies_list]\n    cv_scores = [results[s]['cv_score'] for s in strategies_list]\n    cv_stds = [results[s]['cv_std'] for s in strategies_list]\n    \n    x = np.arange(len(strategies_list))\n    width = 0.35\n    \n    bars1 = ax.bar(x - width/2, train_scores, width, label='Train Score', alpha=0.8)\n    bars2 = ax.bar(x + width/2, cv_scores, width, yerr=cv_stds, \n                    label='CV Score', alpha=0.8, capsize=5)\n    \n    ax.set_xlabel('Strategy')\n    ax.set_ylabel('Score')\n    ax.set_title('Model Performance Comparison')\n    ax.set_xticks(x)\n    ax.set_xticklabels(strategies_list, rotation=45, ha='right')\n    ax.legend()\n    ax.grid(True, alpha=0.3)\n    \n    # 2. Gap analysis\n    ax = axes[1]\n    gaps = [results[s]['gap'] for s in strategies_list]\n    colors = ['red' if g > 0.01 else 'green' for g in gaps]\n    \n    bars = ax.bar(strategies_list, gaps, color=colors, alpha=0.7)\n    ax.axhline(y=0, color='black', linestyle='-', linewidth=0.5)\n    ax.set_xlabel('Strategy')\n    ax.set_ylabel('Train-CV Gap')\n    ax.set_title('Overfitting Analysis (Lower is Better)')\n    ax.set_xticklabels(strategies_list, rotation=45, ha='right')\n    ax.grid(True, alpha=0.3)\n    \n    # Add value labels\n    for bar, gap in zip(bars, gaps):\n        ax.text(bar.get_x() + bar.get_width()/2, bar.get_height() + 0.001,\n                f'{gap:.4f}', ha='center', va='bottom', fontsize=9)\n    \n    plt.tight_layout()\n    plt.savefig('advanced_strategy_comparison.png', dpi=150, bbox_inches='tight')\n    plt.show()\n    \n    # Summary\n    print(\"\\n\" + \"=\"*80)\n    print(\"📊 STRATEGY COMPARISON SUMMARY\")\n    print(\"=\"*80)\n    \n    summary_df = pd.DataFrame(results).T\n    summary_df['gap_reduction'] = summary_df['gap'].min() - summary_df['gap']\n    \n    print(summary_df.round(4))\n    \n    best_strategy = summary_df['cv_score'].idxmax()\n    print(f\"\\n🏆 Best CV Score: {best_strategy}\")\n    \n    lowest_gap = summary_df['gap'].idxmin()\n    print(f\"🎯 Lowest Gap (least overfitting): {lowest_gap}\")\n    \n    return results\n\n# Demonstration of segment-specific transformations\ndef demonstrate_segment_transformations():\n    \"\"\"Show how different segments can have different transformations\"\"\"\n    print(\"\\n\" + \"=\"*80)\n    print(\"📐 DEMONSTRATING SEGMENT-SPECIFIC TRANSFORMATIONS\")\n    print(\"=\"*80)\n    \n    # Create synthetic data with different characteristics in each quartile\n    np.random.seed(42)\n    n_samples = 2000\n    \n    # Q1: Compressed data (needs expansion)\n    q1 = np.random.normal(0, 0.1, n_samples // 4)\n    \n    # Q2: Normal data (minimal change needed)\n    q2 = np.random.normal(1, 0.3, n_samples // 4)\n    \n    # Q3: Outlier-heavy (needs compression)\n    q3 = np.concatenate([\n        np.random.normal(2, 0.2, n_samples // 8),\n        np.random.uniform(1.5, 2.5, n_samples // 8)\n    ])\n    \n    # Q4: Long tail (needs log-like compression)\n    q4 = np.random.exponential(1, n_samples // 4) + 2.5\n    \n    # Combine\n    data = np.concatenate([q1, q2, q3, q4])\n    np.random.shuffle(data)\n    \n    # Create figure\n    fig, axes = plt.subplots(2, 2, figsize=(15, 12))\n    fig.suptitle('Segment-Specific Transformation Examples', fontsize=16)\n    \n    # Define custom transformations for each quartile\n    quartiles = np.percentile(data, [0, 25, 50, 75, 100])\n    \n    transformations = [\n        {\n            'range': (quartiles[0], quartiles[1]),\n            'name': 'Q1: Expansion',\n            'scale': 1.5,  # Expand by 1.5x\n            'curve': 'power',\n            'strength': 0.7\n        },\n        {\n            'range': (quartiles[1], quartiles[2]),\n            'name': 'Q2: Minimal',\n            'scale': 1.0,  # No change\n            'curve': 'linear',\n            'strength': 1.0\n        },\n        {\n            'range': (quartiles[2], quartiles[3]),\n            'name': 'Q3: Compression',\n            'scale': 0.6,  # Compress to 60%\n            'curve': 'sigmoid',\n            'strength': 2.0\n        },\n        {\n            'range': (quartiles[3], quartiles[4]),\n            'name': 'Q4: Log Compression',\n            'scale': 0.4,  # Strong compression\n            'curve': 'log',\n            'strength': 3.0\n        }\n    ]\n    \n    # Visualize each transformation\n    for idx, transform in enumerate(transformations):\n        ax = axes[idx // 2, idx % 2]\n        \n        # Get data in this segment\n        mask = (data >= transform['range'][0]) & (data <= transform['range'][1])\n        segment_data = data[mask]\n        \n        # Apply transformation\n        compressor = BidirectionalQuartileCompressor()\n        transformed = compressor.apply_segment_transformation(\n            segment_data,\n            {'scale': transform['scale'], 'shift': 0, \n             'curve_type': transform['curve'], 'strength': transform['strength']},\n            transform['range']\n        )\n        \n        # Plot\n        ax.hist(segment_data, bins=30, alpha=0.5, label='Original', density=True, color='blue')\n        ax.hist(transformed, bins=30, alpha=0.5, label='Transformed', density=True, color='red')\n        \n        ax.set_title(transform['name'])\n        ax.set_xlabel('Value')\n        ax.set_ylabel('Density')\n        ax.legend()\n        \n        # Add statistics\n        stats_text = f\"Original: μ={np.mean(segment_data):.2f}, σ={np.std(segment_data):.2f}\\n\"\n        stats_text += f\"Transform: μ={np.mean(transformed):.2f}, σ={np.std(transformed):.2f}\\n\"\n        stats_text += f\"Scale: {transform['scale']:.1f}x\"\n        \n        ax.text(0.02, 0.98, stats_text, transform=ax.transAxes,\n                verticalalignment='top', fontsize=9,\n                bbox=dict(boxstyle='round', facecolor='wheat', alpha=0.5))\n    \n    plt.tight_layout()\n    plt.savefig('segment_transformation_examples.png', dpi=150, bbox_inches='tight')\n    plt.show()\n\n# Main execution\nif __name__ == \"__main__\":\n    # Load data\n    print(\"📊 Loading data...\")\n    \n    # For demonstration, create synthetic data\n    np.random.seed(42)\n    n_samples = 5000\n    n_features = 20\n    \n    # Create features with different characteristics\n    X_train = pd.DataFrame()\n    \n    # Some normal features\n    for i in range(5):\n        X_train[f'normal_{i}'] = np.random.normal(0, 1, n_samples)\n    \n    # Some skewed features\n    for i in range(5):\n        X_train[f'skewed_{i}'] = np.random.exponential(1, n_samples)\n    \n    # Some features with outliers\n    for i in range(5):\n        base = np.random.normal(0, 1, n_samples)\n        outliers = np.random.choice(n_samples, size=int(n_samples * 0.05), replace=False)\n        base[outliers] *= 10\n        X_train[f'outlier_{i}'] = base\n    \n    # Some compressed features (low variance)\n    for i in range(5):\n        X_train[f'compressed_{i}'] = np.random.normal(0, 0.1, n_samples)\n    \n    # Create target with some relationship to features\n    y_train = (\n        0.3 * X_train['normal_0'] +\n        0.2 * np.log1p(X_train['skewed_0']) +\n        0.1 * X_train['outlier_0'].clip(-3, 3) +\n        0.4 * np.random.normal(0, 1, n_samples)\n    )\n    \n    print(f\"Dataset shape: {X_train.shape}\")\n    print(f\"Features: {list(X_train.columns)}\")\n    \n    # Demonstrate segment transformations\n    demonstrate_segment_transformations()\n    \n    # Compare strategies\n    results = compare_compression_strategies_advanced(X_train, pd.Series(y_train))\n    \n    # Final summary\n    print(\"\\n\" + \"=\"*80)\n    print(\"🎓 KEY INSIGHTS: BIDIRECTIONAL ADAPTIVE COMPRESSION\")\n    print(\"=\"*80)\n    \n    print(\"\\n1. **Segment-Specific Transformations**\")\n    print(\"   - Different parts of the distribution need different treatments\")\n    print(\"   - Some regions need expansion (compressed data)\")\n    print(\"   - Others need compression (outliers, long tails)\")\n    \n    print(\"\\n2. **Optimization Based on Train-CV Gap**\")\n    print(\"   - Minimizing the gap ensures we're reducing noise, not signal\")\n    print(\"   - Prevents overfitting to training data artifacts\")\n    \n    print(\"\\n3. **Flexible Segmentation**\")\n    print(\"   - Not limited to quartiles - can use any number of segments\")\n    print(\"   - More segments = more flexibility but higher complexity\")\n    \n    print(\"\\n4. **Gradient Curves**\")\n    print(\"   - Smooth transformations prevent discontinuities\")\n    print(\"   - Different curve types (linear, sigmoid, power, log) for different patterns\")\n    \n    print(\"\\n5. **Bidirectional Benefits**\")\n    print(\"   - Can both compress AND expand as needed\")\n    print(\"   - Adapts to the actual data distribution\")\n    print(\"   - More effective than one-size-fits-all compression\")\n    \n    print(\"\\n\" + \"=\"*80)\n    print(\"✅ Advanced compression demonstration complete!\")\n    print(\"=\"*80)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-25T01:44:26.266204Z","iopub.execute_input":"2025-07-25T01:44:26.267204Z","iopub.status.idle":"2025-07-25T01:56:14.256657Z","shell.execute_reply.started":"2025-07-25T01:44:26.267168Z","shell.execute_reply":"2025-07-25T01:56:14.255669Z"}},"outputs":[],"execution_count":null}]}