{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":96164,"databundleVersionId":12993472,"sourceType":"competition"},{"sourceId":12405352,"sourceType":"datasetVersion","datasetId":7823194},{"sourceId":12462463,"sourceType":"datasetVersion","datasetId":7861436},{"sourceId":12503273,"sourceType":"datasetVersion","datasetId":7870836},{"sourceId":12512096,"sourceType":"datasetVersion","datasetId":7897419},{"sourceId":12522757,"sourceType":"datasetVersion","datasetId":7904568},{"sourceId":12558997,"sourceType":"datasetVersion","datasetId":7909478},{"sourceId":12570279,"sourceType":"datasetVersion","datasetId":7937103},{"sourceId":250087140,"sourceType":"kernelVersion"},{"sourceId":250228944,"sourceType":"kernelVersion"},{"sourceId":250269073,"sourceType":"kernelVersion"},{"sourceId":250368902,"sourceType":"kernelVersion"},{"sourceId":251126076,"sourceType":"kernelVersion"},{"sourceId":251597822,"sourceType":"kernelVersion"},{"sourceId":252223128,"sourceType":"kernelVersion"},{"sourceId":252321999,"sourceType":"kernelVersion"},{"sourceId":252647965,"sourceType":"kernelVersion"}],"dockerImageVersionId":31089,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"Credit to the long list of people who contributed to the ensembling/blending chain and those who authored the attached notebooks and datasets.","metadata":{"_uuid":"55e718ac-8324-4dda-a878-860a5c53e82f","_cell_guid":"a9759f18-c1f3-429d-8a8e-0bb3a2f7e9fd","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false}}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nfrom sklearn.cluster import KMeans\nfrom sklearn.preprocessing import StandardScaler\nimport json\nfrom datetime import datetime\nimport warnings\nwarnings.filterwarnings('ignore')\n\nclass IntelligentSegmentWeighting:\n    def __init__(self, path_to_ds, base_files, additional_files):\n        self.path_to_ds = path_to_ds\n        self.base_files = base_files\n        self.additional_files = additional_files\n        self.all_predictions = None\n        self.segments = None\n        self.segment_weights = None\n        self.outlier_mask = None\n        \n    def load_all_predictions(self):\n        \"\"\"Load all prediction files into a single DataFrame\"\"\"\n        print(\"Loading all prediction files...\")\n        predictions_dict = {}\n        \n        # Load base files\n        for i, file_name in enumerate(self.base_files):\n            file_path = self.path_to_ds + file_name + \".csv\"\n            df = pd.read_csv(file_path)\n            predictions_dict[f'base_{i}_{file_name}'] = df['prediction'].values\n            \n        # Load additional files\n        for i, file_path in enumerate(self.additional_files):\n            df = pd.read_csv(file_path)\n            predictions_dict[f'add_{i}'] = df['prediction'].values\n            \n        # Create DataFrame with all predictions\n        self.all_predictions = pd.DataFrame(predictions_dict)\n        self.all_predictions['ID'] = range(1, len(self.all_predictions) + 1)\n        \n        print(f\"Loaded {len(predictions_dict)} prediction files\")\n        return self.all_predictions\n    \n    def analyze_prediction_patterns(self):\n        \"\"\"Analyze agreement, disagreement, and outlier patterns\"\"\"\n        print(\"\\nAnalyzing prediction patterns...\")\n        \n        pred_cols = [col for col in self.all_predictions.columns if col != 'ID']\n        predictions = self.all_predictions[pred_cols].values\n        \n        # Calculate statistics for each row\n        stats = {\n            'mean': np.mean(predictions, axis=1),\n            'std': np.std(predictions, axis=1),\n            'min': np.min(predictions, axis=1),\n            'max': np.max(predictions, axis=1),\n            'range': np.max(predictions, axis=1) - np.min(predictions, axis=1),\n            'cv': np.std(predictions, axis=1) / (np.abs(np.mean(predictions, axis=1)) + 1e-8)\n        }\n        \n        # Calculate agreement metrics\n        base_cols = [col for col in pred_cols if col.startswith('base_')]\n        add_cols = [col for col in pred_cols if col.startswith('add_')]\n        \n        base_predictions = self.all_predictions[base_cols].values\n        add_predictions = self.all_predictions[add_cols].values\n        \n        # Agreement within base models\n        stats['base_agreement'] = 1 - np.std(base_predictions, axis=1) / (np.abs(np.mean(base_predictions, axis=1)) + 1e-8)\n        \n        # Deviation of additional models from base consensus\n        base_mean = np.mean(base_predictions, axis=1)\n        add_mean = np.mean(add_predictions, axis=1)\n        stats['add_deviation'] = np.abs(add_mean - base_mean)\n        \n        # Identify outliers\n        self.identify_outliers(stats)\n        \n        # Add stats to DataFrame\n        for key, values in stats.items():\n            self.all_predictions[f'stat_{key}'] = values\n            \n        return stats\n    \n    def identify_outliers(self, stats):\n        \"\"\"Identify outlier predictions using multiple criteria\"\"\"\n        print(\"\\nIdentifying outliers...\")\n        \n        # Method 1: High coefficient of variation\n        cv_threshold = np.percentile(stats['cv'], 95)\n        outliers_cv = stats['cv'] > cv_threshold\n        \n        # Method 2: Extreme range\n        range_threshold = np.percentile(stats['range'], 95)\n        outliers_range = stats['range'] > range_threshold\n        \n        # Method 3: Additional models deviate significantly from base\n        dev_threshold = np.percentile(stats['add_deviation'], 90)\n        outliers_deviation = stats['add_deviation'] > dev_threshold\n        \n        # Combine outlier indicators\n        self.outlier_mask = outliers_cv | outliers_range | outliers_deviation\n        \n        print(f\"Identified {np.sum(self.outlier_mask)} outliers ({np.mean(self.outlier_mask)*100:.2f}%)\")\n        print(f\"  - High CV: {np.sum(outliers_cv)}\")\n        print(f\"  - High range: {np.sum(outliers_range)}\")\n        print(f\"  - High additional deviation: {np.sum(outliers_deviation)}\")\n        \n        self.all_predictions['is_outlier'] = self.outlier_mask\n        \n    def create_segments(self, n_segments=5):\n        \"\"\"Create segments based on prediction characteristics\"\"\"\n        print(f\"\\nCreating {n_segments} segments based on prediction patterns...\")\n        \n        # Features for segmentation\n        segment_features = [\n            'stat_mean',\n            'stat_std',\n            'stat_cv',\n            'stat_base_agreement',\n            'stat_add_deviation'\n        ]\n        \n        X = self.all_predictions[segment_features].values\n        \n        # Standardize features\n        scaler = StandardScaler()\n        X_scaled = scaler.fit_transform(X)\n        \n        # Perform clustering\n        kmeans = KMeans(n_clusters=n_segments, random_state=42, n_init=10)\n        self.segments = kmeans.fit_predict(X_scaled)\n        \n        self.all_predictions['segment'] = self.segments\n        \n        # Analyze segments\n        self.analyze_segments()\n        \n        return self.segments\n    \n    def analyze_segments(self):\n        \"\"\"Analyze characteristics of each segment\"\"\"\n        print(\"\\nSegment Analysis:\")\n        \n        segment_stats = []\n        for seg in range(max(self.segments) + 1):\n            mask = self.segments == seg\n            seg_data = self.all_predictions[mask]\n            \n            stats = {\n                'segment': seg,\n                'size': len(seg_data),\n                'pct_of_data': len(seg_data) / len(self.all_predictions) * 100,\n                'avg_prediction': seg_data['stat_mean'].mean(),\n                'avg_std': seg_data['stat_std'].mean(),\n                'avg_base_agreement': seg_data['stat_base_agreement'].mean(),\n                'avg_add_deviation': seg_data['stat_add_deviation'].mean(),\n                'pct_outliers': seg_data['is_outlier'].mean() * 100\n            }\n            \n            segment_stats.append(stats)\n            \n            print(f\"\\nSegment {seg}:\")\n            print(f\"  Size: {stats['size']:,} ({stats['pct_of_data']:.1f}%)\")\n            print(f\"  Avg prediction: {stats['avg_prediction']:.4f}\")\n            print(f\"  Avg std: {stats['avg_std']:.4f}\")\n            print(f\"  Base agreement: {stats['avg_base_agreement']:.3f}\")\n            print(f\"  Additional deviation: {stats['avg_add_deviation']:.4f}\")\n            print(f\"  Outliers: {stats['pct_outliers']:.1f}%\")\n            \n        self.segment_stats = pd.DataFrame(segment_stats)\n        \n    def calculate_intelligent_weights(self):\n        \"\"\"Calculate segment-specific weights based on model behavior\"\"\"\n        print(\"\\nCalculating intelligent segment-specific weights...\")\n        \n        pred_cols = [col for col in self.all_predictions.columns if col.startswith(('base_', 'add_'))]\n        base_cols = [col for col in pred_cols if col.startswith('base_')]\n        add_cols = [col for col in pred_cols if col.startswith('add_')]\n        \n        self.segment_weights = {}\n        \n        for seg in range(max(self.segments) + 1):\n            mask = (self.segments == seg) & (~self.outlier_mask)\n            seg_data = self.all_predictions[mask]\n            \n            if len(seg_data) < 100:  # Skip very small segments\n                continue\n                \n            # Calculate performance metrics for each model in this segment\n            base_agreement = seg_data['stat_base_agreement'].mean()\n            add_deviation = seg_data['stat_add_deviation'].mean()\n            \n            # Weight strategy based on segment characteristics\n            if base_agreement > 0.8 and add_deviation > 0.05:\n                # High base agreement, additional models deviate\n                # Reduce additional model weights significantly\n                additional_weight_total = 0.001  # 0.1%\n                weight_strategy = \"conservative\"\n            elif base_agreement < 0.5:\n                # Low base agreement - more uncertainty\n                # Give additional models more weight\n                additional_weight_total = 0.02  # 2%\n                weight_strategy = \"inclusive\"\n            else:\n                # Moderate agreement\n                additional_weight_total = 0.005  # 0.5%\n                weight_strategy = \"balanced\"\n                \n            # Calculate individual weights\n            base_weight_total = 1.0 - additional_weight_total\n            \n            # Original base weights scaled\n            original_base_weights = [0.90, 0.04, 0.03, 0.02, 0.01]\n            base_weights = [w * base_weight_total for w in original_base_weights]\n            \n            # Additional weights - reduce for high-deviation models\n            add_weights = []\n            for add_col in add_cols:\n                model_deviation = np.abs(seg_data[add_col].mean() - seg_data[base_cols].mean().mean())\n                if model_deviation > add_deviation * 1.5:\n                    # This model deviates too much in this segment\n                    weight = additional_weight_total / len(add_cols) * 0.1\n                else:\n                    weight = additional_weight_total / len(add_cols)\n                add_weights.append(weight)\n                \n            self.segment_weights[seg] = {\n                'base_weights': base_weights,\n                'additional_weights': add_weights,\n                'strategy': weight_strategy,\n                'total_additional': additional_weight_total\n            }\n            \n            print(f\"\\nSegment {seg} weights ({weight_strategy}):\")\n            print(f\"  Total additional weight: {additional_weight_total*100:.2f}%\")\n            print(f\"  Base weights: {[f'{w:.4f}' for w in base_weights[:3]]}...\")\n            print(f\"  Add weights: {[f'{w:.6f}' for w in add_weights[:3]]}...\")\n            \n        # Special handling for outliers\n        self.segment_weights['outlier'] = {\n            'base_weights': [0.95, 0.025, 0.015, 0.007, 0.003],  # Even more conservative\n            'additional_weights': [0.0] * len(add_cols),  # No additional models for outliers\n            'strategy': 'outlier_conservative',\n            'total_additional': 0.0\n        }\n        \n    def create_weighted_ensemble(self, output_file='intelligent_ensemble.csv'):\n        \"\"\"Create final ensemble using segment-specific weights\"\"\"\n        print(f\"\\nCreating weighted ensemble...\")\n        \n        pred_cols = [col for col in self.all_predictions.columns if col.startswith(('base_', 'add_'))]\n        base_cols = [col for col in pred_cols if col.startswith('base_')]\n        add_cols = [col for col in pred_cols if col.startswith('add_')]\n        \n        ensemble_predictions = np.zeros(len(self.all_predictions))\n        \n        # Process each segment\n        for seg in range(max(self.segments) + 1):\n            if seg not in self.segment_weights:\n                continue\n                \n            # Get segment mask\n            mask = self.segments == seg\n            seg_indices = np.where(mask)[0]\n            \n            # Get weights for this segment\n            weights = self.segment_weights[seg]\n            all_weights = weights['base_weights'] + weights['additional_weights']\n            \n            # Calculate weighted predictions for this segment\n            seg_predictions = self.all_predictions.iloc[seg_indices][pred_cols].values\n            weighted_preds = np.sum(seg_predictions * all_weights, axis=1)\n            \n            ensemble_predictions[seg_indices] = weighted_preds\n            \n        # Handle outliers\n        outlier_indices = np.where(self.outlier_mask)[0]\n        if len(outlier_indices) > 0:\n            outlier_weights = self.segment_weights['outlier']\n            all_weights = outlier_weights['base_weights'] + outlier_weights['additional_weights']\n            \n            outlier_predictions = self.all_predictions.iloc[outlier_indices][pred_cols].values\n            weighted_outliers = np.sum(outlier_predictions * all_weights, axis=1)\n            \n            ensemble_predictions[outlier_indices] = weighted_outliers\n            \n        # Create final submission\n        submission = pd.DataFrame({\n            'ID': self.all_predictions['ID'],\n            'prediction': ensemble_predictions\n        })\n        \n        submission.to_csv(output_file, index=False)\n        print(f\"\\nEnsemble saved to {output_file}\")\n        \n        # Save detailed analysis\n        self.save_analysis_report()\n        \n        return submission\n    \n    def save_analysis_report(self):\n        \"\"\"Save detailed analysis report\"\"\"\n        report = {\n            'timestamp': datetime.now().isoformat(),\n            'segment_stats': self.segment_stats.to_dict(),\n            'segment_weights': self.segment_weights,\n            'outlier_count': int(np.sum(self.outlier_mask)),\n            'outlier_percentage': float(np.mean(self.outlier_mask) * 100)\n        }\n        \n        with open('intelligent_ensemble_report.json', 'w') as f:\n            json.dump(report, f, indent=2)\n            \n        # Save segment assignments\n        segment_df = pd.DataFrame({\n            'ID': self.all_predictions['ID'],\n            'segment': self.segments,\n            'is_outlier': self.outlier_mask,\n            'base_agreement': self.all_predictions['stat_base_agreement'],\n            'add_deviation': self.all_predictions['stat_add_deviation']\n        })\n        segment_df.to_csv('segment_assignments.csv', index=False)\n        \n    def create_diagnostic_ensembles(self):\n        \"\"\"Create multiple diagnostic ensembles for testing\"\"\"\n        print(\"\\nCreating diagnostic ensembles...\")\n        \n        # 1. Conservative ensemble (minimal additional weight)\n        self.create_custom_ensemble('ensemble_conservative.csv', additional_weight=0.001)\n        \n        # 2. Balanced ensemble (moderate additional weight) \n        self.create_custom_ensemble('ensemble_balanced.csv', additional_weight=0.005)\n        \n        # 3. No additional ensemble (base models only)\n        self.create_custom_ensemble('ensemble_base_only.csv', additional_weight=0.0)\n        \n        # 4. Segment-aware ensemble (using calculated weights)\n        self.create_weighted_ensemble('ensemble_intelligent.csv')\n        \n        print(\"\\nCreated 4 diagnostic ensembles:\")\n        print(\"  1. ensemble_conservative.csv - Minimal additional weight (0.1%)\")\n        print(\"  2. ensemble_balanced.csv - Moderate additional weight (0.5%)\")\n        print(\"  3. ensemble_base_only.csv - Base models only\")\n        print(\"  4. ensemble_intelligent.csv - Segment-specific intelligent weights\")\n        \n    def create_custom_ensemble(self, output_file, additional_weight):\n        \"\"\"Create ensemble with custom uniform weights\"\"\"\n        pred_cols = [col for col in self.all_predictions.columns if col.startswith(('base_', 'add_'))]\n        base_cols = [col for col in pred_cols if col.startswith('base_')]\n        add_cols = [col for col in pred_cols if col.startswith('add_')]\n        \n        # Calculate weights\n        base_weight_total = 1.0 - additional_weight\n        original_base_weights = [0.90, 0.04, 0.03, 0.02, 0.01]\n        base_weights = [w * base_weight_total for w in original_base_weights]\n        \n        if additional_weight > 0:\n            add_weights = [additional_weight / len(add_cols)] * len(add_cols)\n        else:\n            add_weights = [0.0] * len(add_cols)\n            \n        all_weights = base_weights + add_weights\n        \n        # Calculate ensemble\n        predictions = self.all_predictions[pred_cols].values\n        ensemble_predictions = np.sum(predictions * all_weights, axis=1)\n        \n        # Save\n        submission = pd.DataFrame({\n            'ID': self.all_predictions['ID'],\n            'prediction': ensemble_predictions\n        })\n        submission.to_csv(output_file, index=False)\n\n\n# Wrapper function for easy use\ndef run_intelligent_ensemble(path_to_ds, base_files, additional_files, n_segments=5):\n    \"\"\"Run the complete intelligent ensemble process\"\"\"\n    \n    # Initialize system\n    system = IntelligentSegmentWeighting(path_to_ds, base_files, additional_files)\n    \n    # Load all predictions\n    system.load_all_predictions()\n    \n    # Analyze patterns\n    system.analyze_prediction_patterns()\n    \n    # Create segments\n    system.create_segments(n_segments=n_segments)\n    \n    # Calculate intelligent weights\n    system.calculate_intelligent_weights()\n    \n    # Create diagnostic ensembles\n    system.create_diagnostic_ensembles()\n    \n    return system\n\n\n# Example usage\nif __name__ == \"__main__\":\n    # Configuration\n    path_to_ds = '/kaggle/input/24-juli-2025-drw/submission '\n    base_files = ['0.95167', '0.95164', '0.95129', '0.95109', '0.90006']\n    additional_files = [\n        \"/kaggle/input/anyone-can-win-on-public-lb/submission.csv\",\n        \"/kaggle/input/drw-ensembling-0-86767/submission.csv\",\n        \"/kaggle/input/model-of-best-5/submission.csv\",\n        \"/kaggle/input/drw-crypto-market-prediction-iblend/submission.csv\",\n        \"/kaggle/input/drwmy/submission.csv\",\n        \"/kaggle/input/final-solution/submission.csv\",\n        \"/kaggle/input/drw-ensemble-featearue-importance/submission.csv\"\n    ]\n    \n    # Run the intelligent ensemble system\n    system = run_intelligent_ensemble(path_to_ds, base_files, additional_files, n_segments=5)\n    \n    print(\"\\n\" + \"=\"*70)\n    print(\"ENSEMBLE CREATION COMPLETE\")\n    print(\"=\"*70)\n    print(\"\\nNext steps:\")\n    print(\"1. Submit the four diagnostic ensembles to evaluate performance\")\n    print(\"2. Review 'intelligent_ensemble_report.json' for detailed analysis\")\n    print(\"3. Check 'segment_assignments.csv' to understand prediction groupings\")\n    print(\"4. Based on results, adjust segment count or weight strategies\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-24T14:29:25.533439Z","iopub.execute_input":"2025-07-24T14:29:25.534428Z"}},"outputs":[],"execution_count":null}]}